{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# logistic regression & feature selection is all u need\n\nThis is a simple notebook fitting logistic regression Grid Search. \n\nSpecial thanks to Devastator for his [clean notebook ](https://www.kaggle.com/code/thedevastator/tps-aug-simple-baseline)\nand to Lucas See for his [great EDA](https://www.kaggle.com/code/pinstripezebra/eda-baseline-model). Make sure to give them an upvote! ","metadata":{}},{"cell_type":"markdown","source":"![](https://media.slid.es/uploads/1047697/images/6229366/pasted-from-clipboard.png)","metadata":{}},{"cell_type":"code","source":"#%%capture\n#!pip install feature_engine\n#!pip install missingno\n# all may not be needed\n\nimport os\nimport sys\nimport numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport matplotlib.pylab as plt\n\n\nfrom scipy.stats import uniform\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier,ExtraTreesClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.ensemble import StackingClassifier,VotingClassifier,StackingClassifier\n\nfrom catboost import CatBoostClassifier\n\nfrom sklearn.model_selection import cross_validate, StratifiedKFold, RepeatedStratifiedKFold, RandomizedSearchCV, GridSearchCV\nfrom sklearn.metrics import confusion_matrix, plot_confusion_matrix, classification_report, roc_auc_score, accuracy_score\nfrom sklearn import metrics\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import SimpleImputer, KNNImputer, IterativeImputer\nfrom sklearn import preprocessing\nfrom sklearn.preprocessing import OneHotEncoder, RobustScaler, PowerTransformer, LabelEncoder, StandardScaler, MinMaxScaler\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-08T11:17:20.725571Z","iopub.execute_input":"2022-08-08T11:17:20.726158Z","iopub.status.idle":"2022-08-08T11:17:20.746181Z","shell.execute_reply.started":"2022-08-08T11:17:20.726115Z","shell.execute_reply":"2022-08-08T11:17:20.744772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Load\ntrain = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\nsubmission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:21.095572Z","iopub.execute_input":"2022-08-08T11:17:21.096887Z","iopub.status.idle":"2022-08-08T11:17:21.261521Z","shell.execute_reply.started":"2022-08-08T11:17:21.096836Z","shell.execute_reply":"2022-08-08T11:17:21.260660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Variable lists for easy manipulation\nid_var = ['id']\ntarget = train.pop('failure')\ncat_vars = ['product_code','attribute_0','attribute_1']\nnum_vars = [v for v in test.columns if v not in id_var and v not in cat_vars]\nnullValue_cols = [col for col in train.columns if train[col].isnull().sum()!=0]\npredictors = cat_vars + num_vars","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:21.423085Z","iopub.execute_input":"2022-08-08T11:17:21.423754Z","iopub.status.idle":"2022-08-08T11:17:21.443922Z","shell.execute_reply.started":"2022-08-08T11:17:21.423717Z","shell.execute_reply":"2022-08-08T11:17:21.442836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Missing Value Imputation\nThis is taken from DES. and The Mike Jones [notebook](https://www.kaggle.com/code/themikejones/tps-aug-22-votingclassifier ","metadata":{}},{"cell_type":"code","source":"# Track missing values \nfor v in nullValue_cols: \n    train[f'na_{v}']= np.where(train[v].isna()==True, 1,0)\n    test[f'na_{v}']= np.where(test[v].isna()==True, 1,0)\nna_vars = [v for v in train.columns if 'na' in v]","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:21.700390Z","iopub.execute_input":"2022-08-08T11:17:21.700775Z","iopub.status.idle":"2022-08-08T11:17:21.733443Z","shell.execute_reply.started":"2022-08-08T11:17:21.700743Z","shell.execute_reply":"2022-08-08T11:17:21.732569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !rm -r kuma_utils\n!git clone https://github.com/analokmaus/kuma_utils.git","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:21.923243Z","iopub.execute_input":"2022-08-08T11:17:21.924008Z","iopub.status.idle":"2022-08-08T11:17:23.060401Z","shell.execute_reply.started":"2022-08-08T11:17:21.923960Z","shell.execute_reply":"2022-08-08T11:17:23.058672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sys.path.append(\"kuma_utils/\")\nfrom kuma_utils.preprocessing.imputer import LGBMImputer","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:23.063003Z","iopub.execute_input":"2022-08-08T11:17:23.063441Z","iopub.status.idle":"2022-08-08T11:17:23.071017Z","shell.execute_reply.started":"2022-08-08T11:17:23.063406Z","shell.execute_reply":"2022-08-08T11:17:23.069856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train['product_code'].unique())\ndf_A = train[train['product_code']=='A']\ndf_B = train[train['product_code']=='B']\ndf_C = train[train['product_code']=='C']\ndf_D = train[train['product_code']=='D']\ndf_E = train[train['product_code']=='E']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:23.072512Z","iopub.execute_input":"2022-08-08T11:17:23.073025Z","iopub.status.idle":"2022-08-08T11:17:23.119589Z","shell.execute_reply.started":"2022-08-08T11:17:23.072989Z","shell.execute_reply":"2022-08-08T11:17:23.118392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(test['product_code'].unique())\ndf_F_t = test[test['product_code']=='F']\ndf_G_t = test[test['product_code']=='G']\ndf_H_t = test[test['product_code']=='H']\ndf_I_t = test[test['product_code']=='I']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:23.122427Z","iopub.execute_input":"2022-08-08T11:17:23.123133Z","iopub.status.idle":"2022-08-08T11:17:23.150854Z","shell.execute_reply.started":"2022-08-08T11:17:23.123087Z","shell.execute_reply":"2022-08-08T11:17:23.149614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_imtr = LGBMImputer(cat_features=cat_vars, n_iter=50)\n\n# train dataset\ntrain_iterimp_A = lgbm_imtr.fit_transform(df_A[nullValue_cols])\ntrain_iterimp_B = lgbm_imtr.fit_transform(df_B[nullValue_cols])\ntrain_iterimp_C = lgbm_imtr.fit_transform(df_C[nullValue_cols])\ntrain_iterimp_D = lgbm_imtr.fit_transform(df_D[nullValue_cols])\ntrain_iterimp_E = lgbm_imtr.fit_transform(df_E[nullValue_cols])\n\n# test dataset\ntest_iterimp_F = lgbm_imtr.fit_transform(df_F_t[nullValue_cols])\ntest_iterimp_G = lgbm_imtr.fit_transform(df_G_t[nullValue_cols])\ntest_iterimp_H = lgbm_imtr.fit_transform(df_H_t[nullValue_cols])\ntest_iterimp_I = lgbm_imtr.fit_transform(df_I_t[nullValue_cols])","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-08T11:17:23.152469Z","iopub.execute_input":"2022-08-08T11:17:23.152814Z","iopub.status.idle":"2022-08-08T11:17:44.480637Z","shell.execute_reply.started":"2022-08-08T11:17:23.152781Z","shell.execute_reply":"2022-08-08T11:17:44.479641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"none_na_cols = [col for col in train.columns if col not in nullValue_cols]\ndf_train = train[none_na_cols]\ndf_test = test[none_na_cols]\n\ntrain_ = pd.concat([train_iterimp_A, train_iterimp_B,train_iterimp_C,train_iterimp_D,train_iterimp_E], axis=0)\ntrain = pd.concat([df_train, train_], axis=1)\n\ntest_ = pd.concat([test_iterimp_F, test_iterimp_G,test_iterimp_H,test_iterimp_I], axis=0)\ntest = pd.concat([df_test, test_], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:44.485370Z","iopub.execute_input":"2022-08-08T11:17:44.487487Z","iopub.status.idle":"2022-08-08T11:17:44.521716Z","shell.execute_reply.started":"2022-08-08T11:17:44.487448Z","shell.execute_reply":"2022-08-08T11:17:44.520686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Missing values in train dataset after pre-peocessing is: \", format(train.isna().sum().sum()))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:44.523078Z","iopub.execute_input":"2022-08-08T11:17:44.523396Z","iopub.status.idle":"2022-08-08T11:17:44.537408Z","shell.execute_reply.started":"2022-08-08T11:17:44.523366Z","shell.execute_reply":"2022-08-08T11:17:44.536149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering\nThis is taken from DES. and The Mike Jones [notebook](https://www.kaggle.com/code/themikejones/tps-aug-22-votingclassifier ","metadata":{}},{"cell_type":"code","source":"# Combination\ntrain['attribute_2*3'] = train['attribute_2'] * train['attribute_3']\ntest['attribute_2*3'] = test['attribute_2'] * test['attribute_3']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:44.538968Z","iopub.execute_input":"2022-08-08T11:17:44.539770Z","iopub.status.idle":"2022-08-08T11:17:44.549529Z","shell.execute_reply.started":"2022-08-08T11:17:44.539736Z","shell.execute_reply":"2022-08-08T11:17:44.548626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# aggregation\nmeas_gr1_cols = [f\"measurement_{i:d}\" for i in list(range(3, 5)) + list(range(9, 17))]\ntrain['meas_gr1_avg'] = np.mean(train[meas_gr1_cols], axis=1)\ntrain['meas_gr1_std'] = np.std(train[meas_gr1_cols], axis=1)\n\ntest['meas_gr1_avg'] = np.mean(test[meas_gr1_cols], axis=1)\ntest['meas_gr1_std'] = np.std(test[meas_gr1_cols], axis=1) \n\nmeas_gr2_cols = [f\"measurement_{i:d}\" for i in list(range(5, 9))]\ntrain['meas_gr2_avg'] = np.mean(train[meas_gr2_cols], axis=1)\ntest['meas_gr2_avg'] = np.mean(test[meas_gr2_cols], axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:44.551022Z","iopub.execute_input":"2022-08-08T11:17:44.551601Z","iopub.status.idle":"2022-08-08T11:17:44.630654Z","shell.execute_reply.started":"2022-08-08T11:17:44.551552Z","shell.execute_reply":"2022-08-08T11:17:44.629648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ratio\ntrain['meas17/meas_gr2_avg'] = train['measurement_17'] / train['meas_gr2_avg']\ntest['meas17/meas_gr2_avg'] = test['measurement_17'] / test['meas_gr2_avg']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:44.633527Z","iopub.execute_input":"2022-08-08T11:17:44.634049Z","iopub.status.idle":"2022-08-08T11:17:44.640598Z","shell.execute_reply.started":"2022-08-08T11:17:44.634017Z","shell.execute_reply":"2022-08-08T11:17:44.639650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Categorical Encoding","metadata":{}},{"cell_type":"code","source":"# Drop product code\ntest = test.drop(['product_code'], axis = 1)\ntrain = train.drop(['product_code'], axis = 1)\ncat_vars.remove('product_code')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:44.642086Z","iopub.execute_input":"2022-08-08T11:17:44.642696Z","iopub.status.idle":"2022-08-08T11:17:44.663487Z","shell.execute_reply.started":"2022-08-08T11:17:44.642663Z","shell.execute_reply":"2022-08-08T11:17:44.662406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Label Encoding\nlabel_encoder = LabelEncoder()\ntrain_le = train.copy()\ntest_le = test.copy()\n\nfor col in ['attribute_0', 'attribute_1']:\n    train_le[col] = label_encoder.fit_transform(train[col])\n    test_le[col] = label_encoder.fit_transform(test[col]) \n        \ntrain = train_le\ntest = test_le","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:44.665230Z","iopub.execute_input":"2022-08-08T11:17:44.666016Z","iopub.status.idle":"2022-08-08T11:17:44.702790Z","shell.execute_reply.started":"2022-08-08T11:17:44.665972Z","shell.execute_reply":"2022-08-08T11:17:44.701543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# update predictor list\npredictors = [v for v in train.columns if v not in cat_vars and v not in 'id' and v not in 'failure']\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:44.704496Z","iopub.execute_input":"2022-08-08T11:17:44.704953Z","iopub.status.idle":"2022-08-08T11:17:44.713038Z","shell.execute_reply.started":"2022-08-08T11:17:44.704906Z","shell.execute_reply":"2022-08-08T11:17:44.711993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Selection","metadata":{}},{"cell_type":"code","source":"def FisherScore(bt, y_train, predictors):\n    \"\"\"\n    Verbeke, W., Dejaeger, K., Martens, D., Hur, J., & Baesens, B. (2012). New insights\n    into churn prediction in the telecommunication sector: A profit driven data mining\n    approach. European Journal of Operational Research, 218(1), 211-229.\n    \"\"\"\n    \n    # Get the unique values of dependent variable\n    target_var_val = y_train.unique()\n    \n    # Calculate FisherScore for each predictor\n    predictor_FisherScore = []\n    for v in predictors:\n        fs = np.abs(np.mean(bt.loc[y_train == target_var_val[0], v]) - np.mean(bt.loc[y_train == target_var_val[1], v])) / \\\n             np.sqrt(np.var(bt.loc[y_train == target_var_val[0], v]) + np.var(bt.loc[y_train == target_var_val[1], v]))\n        predictor_FisherScore.append(fs)\n    return predictor_FisherScore","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:44.714624Z","iopub.execute_input":"2022-08-08T11:17:44.715228Z","iopub.status.idle":"2022-08-08T11:17:44.724323Z","shell.execute_reply.started":"2022-08-08T11:17:44.715176Z","shell.execute_reply":"2022-08-08T11:17:44.723274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate Fisher Score for all variables\nfs = FisherScore(train, target, predictors)\nfs_df = pd.DataFrame({\"predictor\":predictors, \"fisherscore\":fs})\nfs_df = fs_df.sort_values('fisherscore', ascending=False).reset_index(drop=True)\nfs_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:44.726563Z","iopub.execute_input":"2022-08-08T11:17:44.726925Z","iopub.status.idle":"2022-08-08T11:17:44.892984Z","shell.execute_reply.started":"2022-08-08T11:17:44.726895Z","shell.execute_reply":"2022-08-08T11:17:44.891865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check how AUC changes when adding more variables sorted by fisher score importance\nfs_scores = []\ntop_n_vars = len(fs_df)\nfor i in range(1, top_n_vars+1):\n    if i % 10 == 0: print('Added # top vars :', i)\n    top_n_predictors = fs_df['predictor'][:i]\n    clf = LogisticRegression(max_iter = 2000, C=0.0001, penalty='l2', solver='newton-cg')\n    fs_scores.append(cross_validate(clf, train[top_n_predictors], target,\n                                    scoring='roc_auc', cv=5, verbose=0, n_jobs=-1, return_train_score=True))\n    \n# Plot\nplt.plot([s['train_score'].mean() for s in fs_scores], color='blue')\nplt.plot([s['test_score'].mean() for s in fs_scores], color='red')\nplt.xlabel('# vars')\nplt.ylabel('AUC')\nplt.legend(['train', 'test'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:17:44.894547Z","iopub.execute_input":"2022-08-08T11:17:44.894980Z","iopub.status.idle":"2022-08-08T11:18:50.518408Z","shell.execute_reply.started":"2022-08-08T11:17:44.894938Z","shell.execute_reply":"2022-08-08T11:18:50.517465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select the top variables based on Fisher Score\nn_vars = int(pd.DataFrame([s['test_score'].mean() for s in fs_scores]).idxmax())+1\nprint('Selected number of variables:', n_vars)\ntop_fs_vars = fs_df['predictor'].iloc[:int(n_vars)].to_list()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:24:06.082818Z","iopub.execute_input":"2022-08-08T11:24:06.083974Z","iopub.status.idle":"2022-08-08T11:24:06.095043Z","shell.execute_reply.started":"2022-08-08T11:24:06.083895Z","shell.execute_reply":"2022-08-08T11:24:06.093926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_fs_vars","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:24:07.695390Z","iopub.execute_input":"2022-08-08T11:24:07.695885Z","iopub.status.idle":"2022-08-08T11:24:07.704791Z","shell.execute_reply.started":"2022-08-08T11:24:07.695839Z","shell.execute_reply":"2022-08-08T11:24:07.703544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_to_use = ['attribute_0','measurement_0', 'measurement_1', 'measurement_2', 'attribute_0','na_measurement_3', 'na_measurement_5',\n               'meas_gr1_avg', 'meas_gr1_std', 'attribute_2*3', 'loading', 'measurement_17', 'meas17/meas_gr2_avg']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:22:19.017358Z","iopub.execute_input":"2022-08-08T11:22:19.017752Z","iopub.status.idle":"2022-08-08T11:22:19.024036Z","shell.execute_reply.started":"2022-08-08T11:22:19.017721Z","shell.execute_reply":"2022-08-08T11:22:19.022473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling","metadata":{}},{"cell_type":"code","source":"fold = 5\nseed = 666\ndef score(X, y, model, cv):\n    scoring = [\"roc_auc\"]\n    scores = cross_validate(\n        model, X, y, scoring=scoring, cv=cv, return_train_score=True,\n    )\n    scores = pd.DataFrame(scores).T\n    return scores.assign(\n        mean = lambda x: x.mean(axis=1),\n        std = lambda x: x.std(axis=1),\n    )\n\nskf = StratifiedKFold(n_splits=fold, shuffle=True, random_state=seed)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:20:55.973691Z","iopub.execute_input":"2022-08-08T11:20:55.974215Z","iopub.status.idle":"2022-08-08T11:20:55.982625Z","shell.execute_reply.started":"2022-08-08T11:20:55.974176Z","shell.execute_reply":"2022-08-08T11:20:55.981177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train test split\nX_train, X_test, y_train, y_test = train_test_split(train[predictors], target, test_size=0.20, random_state=42)\nX_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:20:56.751823Z","iopub.execute_input":"2022-08-08T11:20:56.753028Z","iopub.status.idle":"2022-08-08T11:20:56.800137Z","shell.execute_reply.started":"2022-08-08T11:20:56.752984Z","shell.execute_reply":"2022-08-08T11:20:56.798727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using all predictors\nlrgs1 = LogisticRegression(max_iter = 200, C=0.0001, penalty='l2', solver='newton-cg')\nlrgs1.fit(train[predictors], target)\n\nscores1 = score(train[predictors], target, lrgs1, cv=skf)\ndisplay(scores1)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:20:58.035802Z","iopub.execute_input":"2022-08-08T11:20:58.036624Z","iopub.status.idle":"2022-08-08T11:21:05.073109Z","shell.execute_reply.started":"2022-08-08T11:20:58.036581Z","shell.execute_reply":"2022-08-08T11:21:05.071508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using pre-selected columns\nlrgs2 = LogisticRegression(max_iter = 200, C=0.0001, penalty='l2', solver='newton-cg')\nlrgs2.fit(train[cols_to_use], target)\n\nscores2 = score(train[cols_to_use], target, lrgs2, cv=skf)\ndisplay(scores2)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:22:21.000729Z","iopub.execute_input":"2022-08-08T11:22:21.001233Z","iopub.status.idle":"2022-08-08T11:22:26.734620Z","shell.execute_reply.started":"2022-08-08T11:22:21.001195Z","shell.execute_reply":"2022-08-08T11:22:26.733120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using top fisher score columns\nlrgs3 = LogisticRegression(max_iter = 200, C=0.0001, penalty='l2', solver='newton-cg')\nlrgs3.fit(train[top_fs_vars], target)\n\nscores3 = score(train[top_fs_vars], target, lrgs3, cv=skf)\ndisplay(scores3)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:24:16.473672Z","iopub.execute_input":"2022-08-08T11:24:16.474081Z","iopub.status.idle":"2022-08-08T11:24:22.140165Z","shell.execute_reply.started":"2022-08-08T11:24:16.474034Z","shell.execute_reply":"2022-08-08T11:24:22.138954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"# Remove id\n#test = test.drop('id', axis = 1)\n\nsub1 = submission.copy()\nlrgs2.fit(train[top_fs_vars], target)\nsub1.failure = lrgs2.predict_proba(test[top_fs_vars])[:,1]\nsub1.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:04:30.296827Z","iopub.status.idle":"2022-08-08T11:04:30.297251Z","shell.execute_reply.started":"2022-08-08T11:04:30.297013Z","shell.execute_reply":"2022-08-08T11:04:30.297031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- model = LogisticRegression(tol = 1e-4, max_iter=500,random_state=seed)\n\nsearch_space = dict(C=[0.0001, 0.01, 0.1, 1],\n                     penalty=['l2', 'l1'],\n                     solver= ['saga', 'liblinear', 'newton-cg'])\n\nsearch = RandomizedSearchCV(model,\n                            search_space, \n                            random_state=seed,\n                            cv = 5, \n                            scoring='roc_auc')\n\nrand_search = search.fit(X, y)\n\nprint('Best Hyperparameters: %s' % rand_search.best_params_)\nprint(\"Best Estimator: \\n{}\\n\".format(rand_search.best_estimator_))\nprint(\"Best Score: \\n{}\\n\".format(rand_search.best_score_)) -->","metadata":{"execution":{"iopub.status.busy":"2022-08-05T22:45:55.190524Z","iopub.execute_input":"2022-08-05T22:45:55.190921Z","iopub.status.idle":"2022-08-05T22:46:19.212786Z","shell.execute_reply.started":"2022-08-05T22:45:55.19089Z","shell.execute_reply":"2022-08-05T22:46:19.211702Z"}}}]}