{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Tabular Playground Logistic Regression Starter\n#### Fernando Delgado - August 1, 2022","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Data manipulation\nimport pandas as pd \nimport numpy as np\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.feature_selection import mutual_info_classif, VarianceThreshold\n\n# Machine Learning\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\n\n#Scores\nfrom sklearn import metrics\nfrom sklearn.metrics import accuracy_score, roc_auc_score\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import cross_val_score\n\n# Experimental setup\nfrom sklearn.model_selection import KFold,  RepeatedStratifiedKFold, cross_validate, GridSearchCV","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T09:36:17.784741Z","iopub.execute_input":"2022-08-02T09:36:17.785239Z","iopub.status.idle":"2022-08-02T09:36:17.794231Z","shell.execute_reply.started":"2022-08-02T09:36:17.785201Z","shell.execute_reply":"2022-08-02T09:36:17.792584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', 200)\npd.set_option('display.max_rows', 200)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:18.069785Z","iopub.execute_input":"2022-08-02T09:36:18.071326Z","iopub.status.idle":"2022-08-02T09:36:18.077164Z","shell.execute_reply.started":"2022-08-02T09:36:18.071262Z","shell.execute_reply":"2022-08-02T09:36:18.075651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load data \ntrain = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\nsample_submission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:18.353534Z","iopub.execute_input":"2022-08-02T09:36:18.354710Z","iopub.status.idle":"2022-08-02T09:36:18.565888Z","shell.execute_reply.started":"2022-08-02T09:36:18.354669Z","shell.execute_reply":"2022-08-02T09:36:18.564630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pre-Process","metadata":{}},{"cell_type":"code","source":"# Variable lists for easy manipulation\nid_var = ['id']\ntarget= ['failure']\ncat_vars = ['product_code',\n 'attribute_0',\n 'attribute_1',\n 'attribute_2',\n 'attribute_3',\n 'measurement_0',\n 'measurement_1',\n 'measurement_2']\n\nnum_vars = [v for v in test.columns if v not in id_var and v not in cat_vars]\npredictors = cat_vars + num_vars","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:18.641199Z","iopub.execute_input":"2022-08-02T09:36:18.641615Z","iopub.status.idle":"2022-08-02T09:36:18.648951Z","shell.execute_reply.started":"2022-08-02T09:36:18.641579Z","shell.execute_reply":"2022-08-02T09:36:18.647503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split train test \nX_train, X_test, y_train, y_test = train_test_split(train[predictors], train[target], test_size = 0.2, random_state=42)\nX_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:18.927383Z","iopub.execute_input":"2022-08-02T09:36:18.928154Z","iopub.status.idle":"2022-08-02T09:36:18.984398Z","shell.execute_reply.started":"2022-08-02T09:36:18.928111Z","shell.execute_reply":"2022-08-02T09:36:18.983159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fill missing values\nfor v in num_vars:\n    # Train\n    X_train[f'missing_{v}'] = np.where(X_train[v].isna() == True, 1, 0)\n    X_train[v] = X_train[v].fillna(X_train[v].mean())\n    \n    # Validation\n    X_test[f'missing_{v}'] = np.where(X_test[v].isna() == True, 1, 0)\n    X_test[v] = X_test[v].fillna(X_test[v].mean())\n    \n    # Test\n    test[f'missing_{v}'] = np.where(test[v].isna() == True, 1, 0)\n    test[v] = test[v].fillna(test[v].mean())\n\n# missing vars list\nna_vars = [v for v in X_train.columns if 'missing' in v]","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:19.219816Z","iopub.execute_input":"2022-08-02T09:36:19.220473Z","iopub.status.idle":"2022-08-02T09:36:19.300275Z","shell.execute_reply.started":"2022-08-02T09:36:19.220436Z","shell.execute_reply":"2022-08-02T09:36:19.298439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encode categoricals\nenc = OrdinalEncoder()\nenc.fit(pd.concat([X_train[cat_vars].astype(str), X_test[cat_vars].astype(str),  test[cat_vars].astype(str)], axis=0))\n\n# Apply on train, test\nX_train[cat_vars] = enc.transform(X_train[cat_vars].astype(str))\nX_test[cat_vars] = enc.transform(X_test[cat_vars].astype(str))\ntest[cat_vars] = enc.transform(test[cat_vars].astype(str))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:19.512068Z","iopub.execute_input":"2022-08-02T09:36:19.512547Z","iopub.status.idle":"2022-08-02T09:36:19.987611Z","shell.execute_reply.started":"2022-08-02T09:36:19.512509Z","shell.execute_reply":"2022-08-02T09:36:19.986212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print out the final variables\nprint(\"# id_var [\", len(id_var), \"] :\", id_var)\nprint(\"# num_vars [\", len(num_vars), \"] :\", num_vars[:5], \"...\")\nprint(\"# cat_vars [\", len(cat_vars), \"] :\", cat_vars[:5], \"...\")\nprint(\"# na_vars [\", len(na_vars), \"] :\", na_vars[:5], \"...\")\nprint(\"# target_var [\", len(target), \"] :\", target)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:19.989653Z","iopub.execute_input":"2022-08-02T09:36:19.990058Z","iopub.status.idle":"2022-08-02T09:36:19.998110Z","shell.execute_reply.started":"2022-08-02T09:36:19.990022Z","shell.execute_reply":"2022-08-02T09:36:19.997101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check missing value\nprint('Train - # NA: ', X_train.isna().sum().sum())\nprint('Validation - # NA: ', X_test.isna().sum().sum())\nprint('Test - # NA:', test.isna().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:20.229243Z","iopub.execute_input":"2022-08-02T09:36:20.229944Z","iopub.status.idle":"2022-08-02T09:36:20.249463Z","shell.execute_reply.started":"2022-08-02T09:36:20.229892Z","shell.execute_reply":"2022-08-02T09:36:20.248186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Eng","metadata":{"execution":{"iopub.status.busy":"2022-08-01T12:33:55.556312Z","iopub.execute_input":"2022-08-01T12:33:55.557687Z","iopub.status.idle":"2022-08-01T12:33:55.562376Z","shell.execute_reply.started":"2022-08-01T12:33:55.557638Z","shell.execute_reply":"2022-08-01T12:33:55.561187Z"}}},{"cell_type":"code","source":"# Categorical remapping\ntrans_vars = []\nfor v in cat_vars:\n    # Find the best decision tree using CV\n    cv = KFold(n_splits=5, random_state=1, shuffle=True)\n    model = DecisionTreeClassifier()\n    parameters = {'min_samples_leaf':(X_train.shape[0]*np.array([0.01, 0.025, 0.05, 0.1, 0.25, 0.5])).astype(int)}\n    clf = GridSearchCV(model, parameters, scoring=\"roc_auc\", n_jobs=-1, cv=cv, verbose=0)\n    clf.fit(X_train[[v]], y_train)\n    # Remap the variable on train, test\n    if (clf.best_score_ > 0.5) & (clf.best_estimator_.get_n_leaves() > 1):\n        print(\"Remapping variable\", v,\n                \"from\", X_train[[v]].nunique().values[0],\n                \"to\", clf.best_estimator_.get_n_leaves(), \"categories\")\n        remap_var = v + '_remap'\n        trans_vars.append(remap_var)\n        X_train[remap_var] = [np.nonzero(r)[0].max() for r in clf.best_estimator_.decision_path(X_train[[v]]).toarray()]\n        X_test[remap_var] = [np.nonzero(r)[0].max() for r in clf.best_estimator_.decision_path(X_test[[v]]).toarray()]\n        test[remap_var] = [np.nonzero(r)[0].max() for r in clf.best_estimator_.decision_path(test[[v]]).toarray()]","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:20.505093Z","iopub.execute_input":"2022-08-02T09:36:20.505523Z","iopub.status.idle":"2022-08-02T09:36:25.790949Z","shell.execute_reply.started":"2022-08-02T09:36:20.505487Z","shell.execute_reply":"2022-08-02T09:36:25.789655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# update predictors\npredictors = num_vars + cat_vars + na_vars\npredictors","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:25.793113Z","iopub.execute_input":"2022-08-02T09:36:25.793850Z","iopub.status.idle":"2022-08-02T09:36:25.803203Z","shell.execute_reply.started":"2022-08-02T09:36:25.793781Z","shell.execute_reply":"2022-08-02T09:36:25.801641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Scale numericals\nfrom sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()\nX_train[num_vars] = scaler.fit_transform(X_train[num_vars])\nX_test[num_vars] = scaler.transform(X_test[num_vars])\ntest[num_vars] = scaler.transform(test[num_vars])","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:25.805434Z","iopub.execute_input":"2022-08-02T09:36:25.806696Z","iopub.status.idle":"2022-08-02T09:36:25.845404Z","shell.execute_reply.started":"2022-08-02T09:36:25.806653Z","shell.execute_reply":"2022-08-02T09:36:25.844090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get dummies on categoricals\nX_train = pd.get_dummies(X_train, columns = trans_vars, drop_first = True)\nX_test = pd.get_dummies(X_test, columns = trans_vars, drop_first = True)\ntest = pd.get_dummies(test, columns = trans_vars, drop_first = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:25.848641Z","iopub.execute_input":"2022-08-02T09:36:25.850321Z","iopub.status.idle":"2022-08-02T09:36:25.914813Z","shell.execute_reply.started":"2022-08-02T09:36:25.850259Z","shell.execute_reply":"2022-08-02T09:36:25.913588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape)\nprint(X_test.shape)\nprint(test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:25.916540Z","iopub.execute_input":"2022-08-02T09:36:25.917327Z","iopub.status.idle":"2022-08-02T09:36:25.925315Z","shell.execute_reply.started":"2022-08-02T09:36:25.917287Z","shell.execute_reply":"2022-08-02T09:36:25.923935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fix predictors list\npredictors = test.columns.tolist()\npredictors.remove('id')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:25.927011Z","iopub.execute_input":"2022-08-02T09:36:25.927364Z","iopub.status.idle":"2022-08-02T09:36:25.935581Z","shell.execute_reply.started":"2022-08-02T09:36:25.927332Z","shell.execute_reply":"2022-08-02T09:36:25.934211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Selection","metadata":{}},{"cell_type":"code","source":"def FisherScore(bt, y_train, predictors):\n    \"\"\"\n    This function calculate the Fisher score of a variable.\n\n    Ref:\n    ---\n    Verbeke, W., Dejaeger, K., Martens, D., Hur, J., & Baesens, B. (2012). New insights\n    into churn prediction in the telecommunication sector: A profit driven data mining\n    approach. European Journal of Operational Research, 218(1), 211-229.\n    \"\"\"\n    \n    # Get the unique values of dependent variable\n    target_var_val = y_train.unique()\n    # Calculate FisherScore for each predictor\n    predictor_FisherScore = []\n    for v in predictors:\n        fs = np.abs(np.mean(bt.loc[y_train == target_var_val[0], v]) - np.mean(bt.loc[y_train == target_var_val[1], v])) / \\\n             np.sqrt(np.var(bt.loc[y_train == target_var_val[0], v]) + np.var(bt.loc[y_train == target_var_val[1], v]))\n        predictor_FisherScore.append(fs)\n    return predictor_FisherScore","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:25.937121Z","iopub.execute_input":"2022-08-02T09:36:25.938139Z","iopub.status.idle":"2022-08-02T09:36:25.949365Z","shell.execute_reply.started":"2022-08-02T09:36:25.938096Z","shell.execute_reply":"2022-08-02T09:36:25.948171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate Fisher Score for all variable\nfs = FisherScore(X_train, y_train['failure'], predictors)\nfs_df = pd.DataFrame({\"predictor\":predictors, \"fisherscore\":fs})\nfs_df = fs_df.sort_values('fisherscore', ascending=False)\nfs_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:25.950611Z","iopub.execute_input":"2022-08-02T09:36:25.951001Z","iopub.status.idle":"2022-08-02T09:36:26.287716Z","shell.execute_reply.started":"2022-08-02T09:36:25.950959Z","shell.execute_reply":"2022-08-02T09:36:26.286358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the Fisher Score\nplt.plot(fs_df['fisherscore'].values.squeeze())\nplt.axvline(x=20, linestyle='dashed', color='red')\nplt.xticks(rotation=45)\nplt.xlabel(str(fs_df.shape[0]) + ' predictors')\nplt.ylabel('Fisher Score')\nplt.legend(['Fisher Score', 'Top20'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:36:26.289464Z","iopub.execute_input":"2022-08-02T09:36:26.289983Z","iopub.status.idle":"2022-08-02T09:36:26.516096Z","shell.execute_reply.started":"2022-08-02T09:36:26.289935Z","shell.execute_reply":"2022-08-02T09:36:26.514428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check how AUC change when add more variables: Top n vars\nfs_scores = []\ntop_n_vars = 80\nfor i in range(1, top_n_vars+1):\n    if i % 10 == 0: print('Added # top vars :', i)\n    top_n_predictors = fs_df['predictor'][:i]\n    clf = LogisticRegression(max_iter = 1500)\n    fs_scores.append(cross_validate(clf, X_train[top_n_predictors], y_train.values.squeeze(),\n                                    scoring='roc_auc', cv=5, verbose=0, n_jobs=-1, return_train_score=True))\n\n# How the AUC curve looks like when adding top vars\nplt.plot([s['train_score'].mean() for s in fs_scores], color='blue')\nplt.plot([s['test_score'].mean() for s in fs_scores], color='red')\nplt.xlabel('# vars')\nplt.ylabel('AUC')\nplt.legend(['train', 'test'])\nplt.show()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-02T09:36:26.522173Z","iopub.execute_input":"2022-08-02T09:36:26.523225Z","iopub.status.idle":"2022-08-02T09:40:10.029550Z","shell.execute_reply.started":"2022-08-02T09:36:26.523168Z","shell.execute_reply":"2022-08-02T09:40:10.028466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create df with scores\ntmp = pd.DataFrame(fs_df['predictor'].reset_index(drop=True).loc[:79])\ntmp['train_score'] = [s['train_score'].mean() for s in fs_scores]\ntmp['test_score'] = [s['test_score'].mean() for s in fs_scores]\n\n# remove predictors that worsen score\ntmp['score_worsened'] = np.where(tmp['test_score'].shift(1)>tmp['test_score'], 1,0)\nnew_predictors = tmp[tmp['score_worsened']== 0]","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:10.031283Z","iopub.execute_input":"2022-08-02T09:40:10.031801Z","iopub.status.idle":"2022-08-02T09:40:10.049497Z","shell.execute_reply.started":"2022-08-02T09:40:10.031750Z","shell.execute_reply":"2022-08-02T09:40:10.047938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check how AUC again for new predictors\nfs_scores = []\ntop_n_vars = len(new_predictors)\nfor i in range(1, top_n_vars+1):\n    if i % 10 == 0: print('Added # top vars :', i)\n    top_n_predictors = new_predictors['predictor'][:i]\n    clf = LogisticRegression(max_iter = 1500)\n    fs_scores.append(cross_validate(clf, X_train[top_n_predictors], y_train.values.squeeze(),\n                                    scoring='roc_auc', cv=5, verbose=0, n_jobs=-1, return_train_score=True))\n\n# How the AUC curve looks like when adding top vars\nplt.plot([s['train_score'].mean() for s in fs_scores], color='blue')\nplt.plot([s['test_score'].mean() for s in fs_scores], color='red')\nplt.xlabel('# vars')\nplt.ylabel('AUC')\nplt.legend(['train', 'test'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:10.051291Z","iopub.execute_input":"2022-08-02T09:40:10.051685Z","iopub.status.idle":"2022-08-02T09:40:20.835843Z","shell.execute_reply.started":"2022-08-02T09:40:10.051651Z","shell.execute_reply":"2022-08-02T09:40:20.834250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create df with scores\ntmp = pd.DataFrame(new_predictors['predictor'].reset_index(drop=True).loc[:79])\ntmp['train_score'] = [s['train_score'].mean() for s in fs_scores]\ntmp['test_score'] = [s['test_score'].mean() for s in fs_scores]\n\n# remove predictors that worsen score\ntmp['score_worsened'] = np.where(tmp['test_score'].shift(1)>tmp['test_score'], 1,0)\nnew_predictors = tmp[tmp['score_worsened']== 0]\nlen(new_predictors)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:20.837838Z","iopub.execute_input":"2022-08-02T09:40:20.838268Z","iopub.status.idle":"2022-08-02T09:40:20.854371Z","shell.execute_reply.started":"2022-08-02T09:40:20.838231Z","shell.execute_reply":"2022-08-02T09:40:20.853183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check how AUC again\nfs_scores = []\ntop_n_vars = len(new_predictors)\nfor i in range(1, top_n_vars+1):\n    if i % 10 == 0: print('Added # top vars :', i)\n    top_n_predictors = new_predictors['predictor'][:i]\n    clf = LogisticRegression(max_iter = 1500)\n    fs_scores.append(cross_validate(clf, X_train[top_n_predictors], y_train.values.squeeze(),\n                                    scoring='roc_auc', cv=5, verbose=0, n_jobs=-1, return_train_score=True))\n\n# How the AUC curve looks like when adding top vars\nplt.plot([s['train_score'].mean() for s in fs_scores], color='blue')\nplt.plot([s['test_score'].mean() for s in fs_scores], color='red')\nplt.xlabel('# vars')\nplt.ylabel('AUC')\nplt.legend(['train', 'test'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:20.856215Z","iopub.execute_input":"2022-08-02T09:40:20.856743Z","iopub.status.idle":"2022-08-02T09:40:24.477122Z","shell.execute_reply.started":"2022-08-02T09:40:20.856695Z","shell.execute_reply":"2022-08-02T09:40:24.475482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select the top variables based on Fisher Score\nn_top_fs_vars = len(new_predictors)  # Top FS vars\ntop_fs_vars = fs_df['predictor'].values[:n_top_fs_vars]\nprint(\"Selected # vars :\", len(top_fs_vars))\ntop_fs_vars","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:24.478620Z","iopub.execute_input":"2022-08-02T09:40:24.479157Z","iopub.status.idle":"2022-08-02T09:40:24.490871Z","shell.execute_reply.started":"2022-08-02T09:40:24.479118Z","shell.execute_reply":"2022-08-02T09:40:24.489459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select top vars or all predictors\nused_vars = top_fs_vars","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:24.492420Z","iopub.execute_input":"2022-08-02T09:40:24.492810Z","iopub.status.idle":"2022-08-02T09:40:24.501466Z","shell.execute_reply.started":"2022-08-02T09:40:24.492777Z","shell.execute_reply":"2022-08-02T09:40:24.500267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def performance(model, printname):\n    \"\"\"\n    This function returns the performance of AUC and Accuracy for the specified model.\n    \"model\" represents the name of the model function, and \"printname\" is a string to give the model a name in the overview table. \n    \"\"\"\n    np.random.seed(1)\n    #Train\n    result = {}\n    predictions   = model.predict(X_train[used_vars])\n    probabilities = pd.DataFrame(model.predict_proba(X_train[used_vars]))[1]\n    accuracy      = accuracy_score(y_train, predictions)\n    auc           = roc_auc_score(np.array(y_train),np.array(probabilities))\n    result[str(printname)] = {\"Train_Accuracy\":accuracy, \"Train_AUC\":auc}\n    train_perf = pd.DataFrame(result).transpose()\n\n    #Test\n    result = {}\n    predictions   = model.predict(X_test[used_vars])\n    probabilities = pd.DataFrame(model.predict_proba(X_test[used_vars]))[1]\n    accuracy      = accuracy_score(y_test, predictions)\n    auc           = roc_auc_score(np.array(y_test),np.array(probabilities))\n    result[str(printname)] = {\"Test_Accuracy\":accuracy,\"Test_AUC\":auc}\n    test_perf = pd.DataFrame(result).transpose()\n\n    #Compare Train and Test AUC\n    perf_total = train_perf.merge(test_perf, left_index=True, right_index=True)\n    return(perf_total)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:24.502837Z","iopub.execute_input":"2022-08-02T09:40:24.503218Z","iopub.status.idle":"2022-08-02T09:40:24.516504Z","shell.execute_reply.started":"2022-08-02T09:40:24.503185Z","shell.execute_reply":"2022-08-02T09:40:24.514656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Logistic Regression","metadata":{}},{"cell_type":"code","source":"# Fit simple Logistic Regression\nnp.random.seed(1)\nlr=LogisticRegression(max_iter=2500).fit(X_train[used_vars], y_train.values.squeeze())\nlr_perf = performance(lr, 'Logistic Regression')\nlr_perf","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:24.518308Z","iopub.execute_input":"2022-08-02T09:40:24.519037Z","iopub.status.idle":"2022-08-02T09:40:25.058236Z","shell.execute_reply.started":"2022-08-02T09:40:24.518997Z","shell.execute_reply":"2022-08-02T09:40:25.056960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit simple Logistic Regression all predictors\nnp.random.seed(1)\n\n#update vars\nused_vars = predictors\n\n# fit model\nlr=LogisticRegression(penalty = 'l2', max_iter = 2500).fit(X_train[used_vars], y_train.values.squeeze())\nlr_perf2 = performance(lr, 'Logistic Regression all Predictors')\nlr_perf2","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:25.059919Z","iopub.execute_input":"2022-08-02T09:40:25.060674Z","iopub.status.idle":"2022-08-02T09:40:27.557755Z","shell.execute_reply.started":"2022-08-02T09:40:25.060627Z","shell.execute_reply":"2022-08-02T09:40:27.556420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Grid search with CV\nnp.random.seed(1)\n\n# Only top vars\nused_vars = top_fs_vars\n\n#Set a Parameter Space\n#space={\"C\":np.logspace(-4, 4, 50),\n#     \"penalty\":[\"l2\"]} # l1 lasso l2 ridge\n\n# Grid Search\n#gs=GridSearchCV(LogisticRegression(max_iter=2500), space, cv=10, scoring='roc_auc')\n\n# Fit Model\n#lr_gs = gs.fit(X_train[used_vars],y_train.values.squeeze())\n\n#Show performance\n#print('Config: %s' % lr_gs.best_params_)\n#lr_gs_perf = performance(lr_gs, 'Logistic Regression GS')\n#lr_gs_perf","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:27.559872Z","iopub.execute_input":"2022-08-02T09:40:27.560834Z","iopub.status.idle":"2022-08-02T09:40:27.567958Z","shell.execute_reply.started":"2022-08-02T09:40:27.560768Z","shell.execute_reply":"2022-08-02T09:40:27.566388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit Grid Search Logistic Regression\nnp.random.seed(1)\nlr=LogisticRegression(C = 3237.45754281764, penalty = 'l2', max_iter = 2500).fit(X_train[used_vars], y_train.values.squeeze())\nlr_gs_perf = performance(lr, 'Logistic Regression GS')\nlr_gs_perf","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:27.574186Z","iopub.execute_input":"2022-08-02T09:40:27.575551Z","iopub.status.idle":"2022-08-02T09:40:27.987084Z","shell.execute_reply.started":"2022-08-02T09:40:27.575492Z","shell.execute_reply":"2022-08-02T09:40:27.985614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check Performance\nperf= pd.concat([lr_gs_perf,lr_perf2,lr_perf]).sort_values(by='Test_AUC', ascending=False)\nperf","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:27.993794Z","iopub.execute_input":"2022-08-02T09:40:27.997347Z","iopub.status.idle":"2022-08-02T09:40:28.025399Z","shell.execute_reply.started":"2022-08-02T09:40:27.997277Z","shell.execute_reply":"2022-08-02T09:40:28.024042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create new training set \nnew_train = pd.concat([X_train[used_vars], X_test[used_vars]])\nnew_target = pd.concat([y_train, y_test])","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:28.032052Z","iopub.execute_input":"2022-08-02T09:40:28.035761Z","iopub.status.idle":"2022-08-02T09:40:28.057340Z","shell.execute_reply.started":"2022-08-02T09:40:28.035692Z","shell.execute_reply":"2022-08-02T09:40:28.055917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def new_perf(classifier, table):\n    \"\"\"\n    This function returns the performance of AUC and Accuracy for the specified model.\n    \"model\" represents the name of the model function, and \"printname\" is a string to give the model a name in the overview table. \n    \"\"\"\n    np.random.seed(1)\n\n    probabilities = pd.DataFrame(classifier.predict_proba(new_train))[1]\n    auc           = roc_auc_score(np.array(new_target),np.array(probabilities))\n    table['New_Train_AUC'] = auc\n    return table \n","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:28.068408Z","iopub.execute_input":"2022-08-02T09:40:28.072150Z","iopub.status.idle":"2022-08-02T09:40:28.083807Z","shell.execute_reply.started":"2022-08-02T09:40:28.072081Z","shell.execute_reply":"2022-08-02T09:40:28.082504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Logistic Regression \nlr = LogisticRegression(C = 3237.45754281764, penalty = 'l2', max_iter = 2500).fit(X_train[used_vars], y_train.values.squeeze())\nlr_perf = new_perf(lr, lr_gs_perf)\nlr_perf","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:40:28.085939Z","iopub.execute_input":"2022-08-02T09:40:28.086411Z","iopub.status.idle":"2022-08-02T09:40:28.525906Z","shell.execute_reply.started":"2022-08-02T09:40:28.086364Z","shell.execute_reply":"2022-08-02T09:40:28.524214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict and submit\nsub1 = sample_submission.copy()\nsub1.failure = lr.predict_proba(test[used_vars])[:, 1]\nsub1.sort_values(by=['id'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:52:42.288311Z","iopub.execute_input":"2022-08-02T09:52:42.288853Z","iopub.status.idle":"2022-08-02T09:52:42.310343Z","shell.execute_reply.started":"2022-08-02T09:52:42.288794Z","shell.execute_reply":"2022-08-02T09:52:42.308886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Ensemble","metadata":{}},{"cell_type":"code","source":"sub2 = pd.read_csv('../input/tps-aug-22-eda-logistic-regression-baseline/submission.csv')\nsub2.sort_values(by=['id'], inplace=True)\n\nsub3 = pd.read_csv('../input/tps-aug-simple-baseline/submission.csv')\nsub3.sort_values(by=['id'], inplace=True)\n\nsub4 = pd.read_csv('../input/tps-8-eda-pytorch-baseline/submission.csv')\nsub4.sort_values(by=['id'], inplace=True)\n\nsub5 = pd.read_csv('../input/tps-eda-feature-engineering-selection-more/firstSubmission.csv')\nsub5.sort_values(by=['id'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:52:59.957295Z","iopub.execute_input":"2022-08-02T09:52:59.957723Z","iopub.status.idle":"2022-08-02T09:53:00.017081Z","shell.execute_reply.started":"2022-08-02T09:52:59.957689Z","shell.execute_reply":"2022-08-02T09:53:00.015711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = sample_submission.copy()\nsub.sort_values(by=['id'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:53:08.242844Z","iopub.execute_input":"2022-08-02T09:53:08.243283Z","iopub.status.idle":"2022-08-02T09:53:08.253051Z","shell.execute_reply.started":"2022-08-02T09:53:08.243248Z","shell.execute_reply":"2022-08-02T09:53:08.251763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b = 1000.0\nS = 0.58511\nq = 0.0","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:56:27.871865Z","iopub.execute_input":"2022-08-02T09:56:27.872324Z","iopub.status.idle":"2022-08-02T09:56:27.878023Z","shell.execute_reply.started":"2022-08-02T09:56:27.872281Z","shell.execute_reply":"2022-08-02T09:56:27.876729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['failure'] = sub1['failure']*np.exp(b*(0.58012-S))\nq = q + np.exp(b*(0.58012-S))\nsub['failure'] = sub['failure'] + sub2['failure']*np.exp(b*(0.58511-S))\nq = q + np.exp(b*(0.58511-S))\nsub['failure'] = sub['failure'] + sub3['failure']*np.exp(b*(0.58497-S))\nq = q + np.exp(b*(0.58497-S))\nsub['failure'] = sub['failure'] + sub4['failure']*np.exp(b*(0.58428-S))\nq = q + np.exp(b*(0.58428-S))\nsub['failure'] = sub['failure'] + sub5['failure']*np.exp(b*(0.5837-S))\nq = q + np.exp(b*(0.5837-S))\n\nsub['failure'] = sub['failure']/q\n\n\nprint(q)\nsub.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:56:59.059279Z","iopub.execute_input":"2022-08-02T09:56:59.059760Z","iopub.status.idle":"2022-08-02T09:56:59.086537Z","shell.execute_reply.started":"2022-08-02T09:56:59.059720Z","shell.execute_reply":"2022-08-02T09:56:59.085249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T09:57:28.946430Z","iopub.execute_input":"2022-08-02T09:57:28.946856Z","iopub.status.idle":"2022-08-02T09:57:29.002775Z","shell.execute_reply.started":"2022-08-02T09:57:28.946805Z","shell.execute_reply":"2022-08-02T09:57:29.001561Z"},"trusted":true},"execution_count":null,"outputs":[]}]}