{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Amex default prediction with LGBM\n\n## --  In comparision with Logistic Regression and Decision Tree","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\npd.set_option('display.max_rows', 100)\npd.set_option('display.max_columns', 100)\npd.set_option('display.width', 1000)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-17T00:53:58.618870Z","iopub.execute_input":"2022-09-17T00:53:58.619515Z","iopub.status.idle":"2022-09-17T00:53:58.647552Z","shell.execute_reply.started":"2022-09-17T00:53:58.619464Z","shell.execute_reply":"2022-09-17T00:53:58.645692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"# Import submission\nsub = pd.read_csv(\"../input/submit-file/submission_lgbm_2022-09-16-23_52.csv\")\nsub.to_csv(f\"submission_lgbm.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:53:58.651351Z","iopub.execute_input":"2022-09-17T00:53:58.651986Z","iopub.status.idle":"2022-09-17T00:54:04.411041Z","shell.execute_reply.started":"2022-09-17T00:53:58.651934Z","shell.execute_reply":"2022-09-17T00:54:04.409560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# raise SystemExit(\"Stop right there!\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.412686Z","iopub.execute_input":"2022-09-17T00:54:04.413059Z","iopub.status.idle":"2022-09-17T00:54:04.427689Z","shell.execute_reply.started":"2022-09-17T00:54:04.413025Z","shell.execute_reply":"2022-09-17T00:54:04.421580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sns\nimport gc\n# import cudf\nimport joblib\nimport time\nimport datetime\nfrom tqdm.auto import tqdm\n# from datetime import datetime \nfrom sklearn import metrics\nfrom sklearn.decomposition import PCA\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import GridSearchCV, RandomizedSearchCV\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import HistGradientBoostingClassifier\nfrom sklearn.metrics import roc_auc_score,roc_curve, auc\nfrom sklearn.metrics import confusion_matrix, precision_score, recall_score\nimport lightgbm as lgb\n\n# print (cudf.__version__)\n# conda create -n rapids-22.08 -c rapidsai -c nvidia -c conda-forge \\ cudf=22.08 python=3.9 cudatoolkit=11.5\n# conda install -c rapidsai -c nvidia -c numba -c conda-forge \\ cudf=22.06 python=3.9 cudatoolkit=11.0","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.429017Z","iopub.status.idle":"2022-09-17T00:54:04.429708Z","shell.execute_reply.started":"2022-09-17T00:54:04.429473Z","shell.execute_reply":"2022-09-17T00:54:04.429501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load train data","metadata":{}},{"cell_type":"code","source":"train_X = pd.read_feather(\"../input/amex-feather-fe/train_fe.feather\")\nlabel = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")\nlabel.sort_values(by = \"customer_ID\",ascending=False)\ntrain_X.sort_values(by = \"customer_ID\",ascending=False)\ny = label.drop(\"customer_ID\",axis = 1)[[\"target\"]]\n\nfeatures = [col for col in train_X.columns if col != \"customer_ID\"]\ncat_features = train_X.select_dtypes(include=['category']).columns.to_list()\nprint(train_X.shape)\ntrain_X.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.431506Z","iopub.status.idle":"2022-09-17T00:54:04.432248Z","shell.execute_reply.started":"2022-09-17T00:54:04.432003Z","shell.execute_reply":"2022-09-17T00:54:04.432030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X.isnull().any(axis=0).sum()\n# turn the rest NA to 0\nfor col in tqdm(features):\n    train_X[col] = train_X[col].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.434547Z","iopub.status.idle":"2022-09-17T00:54:04.435003Z","shell.execute_reply.started":"2022-09-17T00:54:04.434791Z","shell.execute_reply":"2022-09-17T00:54:04.434811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# standardize\n# scaler = StandardScaler()\n# scaler.fit(train_X)\n# train_X_std = scaler.transform(train_X)\n# train_X_std.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.437197Z","iopub.status.idle":"2022-09-17T00:54:04.437652Z","shell.execute_reply.started":"2022-09-17T00:54:04.437447Z","shell.execute_reply":"2022-09-17T00:54:04.437467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize = (10,6))\n# train_X[label['target']==1]['D_39_last'].hist(label = 'target = 1', bins = 30, alpha = 0.5, color = 'green')\n# train_X[label['target']==0]['D_39_last'].hist(label = 'target = 0', bins = 30, alpha = 0.5, color = 'red')\n\n# plt.legend()\n# plt.xlabel('D_39 last month')\n# plt.ylabel('Counts')\n# # plt.show()\n# plt.savefig(\"D_39 last.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.440719Z","iopub.status.idle":"2022-09-17T00:54:04.441147Z","shell.execute_reply.started":"2022-09-17T00:54:04.440938Z","shell.execute_reply":"2022-09-17T00:54:04.440956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize = (10,6))\n# train_X[label['target']==1]['D_39_max'].hist(label = 'target = 1', bins = 30, alpha = 0.5, color = 'green')\n# train_X[label['target']==0]['D_39_max'].hist(label = 'target = 0', bins = 30, alpha = 0.5, color = 'red')\n\n# plt.legend()\n# plt.xlabel('D_39 max')\n# plt.ylabel('Counts')\n# # plt.show()\n# plt.savefig(\"D_39 max.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.442030Z","iopub.status.idle":"2022-09-17T00:54:04.442490Z","shell.execute_reply.started":"2022-09-17T00:54:04.442242Z","shell.execute_reply":"2022-09-17T00:54:04.442271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize = (10,6))\n# train_X[label['target']==1]['P_2_mean'].hist(label = 'target = 1', bins = 30, alpha = 0.5, color = 'green')\n# train_X[label['target']==0]['P_2_mean'].hist(label = 'target = 0', bins = 30, alpha = 0.5, color = 'red')\n\n# plt.legend()\n# plt.xlabel('P_2 mean')\n# plt.ylabel('Counts')\n# # plt.show()\n# plt.savefig(\"P_2 mean.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.444258Z","iopub.status.idle":"2022-09-17T00:54:04.444697Z","shell.execute_reply.started":"2022-09-17T00:54:04.444493Z","shell.execute_reply":"2022-09-17T00:54:04.444517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize = (10,6))\n# train_X[label['target']==1]['D_39_diff1'].hist(label = 'target = 1', bins = 30, alpha = 0.5, color = 'green')\n# train_X[label['target']==0]['D_39_diff1'].hist(label = 'target = 0', bins = 30, alpha = 0.5, color = 'red')\n\n# plt.legend()\n# plt.xlabel('D_39 last month diffrence')\n# plt.ylabel('Counts')\n# # plt.show()\n# plt.savefig(\"D_39 diff.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.445938Z","iopub.status.idle":"2022-09-17T00:54:04.446582Z","shell.execute_reply.started":"2022-09-17T00:54:04.446338Z","shell.execute_reply":"2022-09-17T00:54:04.446358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train test split","metadata":{}},{"cell_type":"code","source":"# Split train data into train&split\nX_tra,X_tes,y_tra,y_tes = train_test_split(train_X[features],y,stratify=y,test_size=0.25, random_state=0)\n\n# del train_X, y\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.448252Z","iopub.status.idle":"2022-09-17T00:54:04.448717Z","shell.execute_reply.started":"2022-09-17T00:54:04.448518Z","shell.execute_reply":"2022-09-17T00:54:04.448538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Define amex metric","metadata":{}},{"cell_type":"code","source":"# Define amex metric\n\ndef amex_metric2(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)\n\ndef amex_metric(y_true, y_pred):\n    labels = np.transpose(np.array([y_true, y_pred]))\n    labels = labels[labels[:, 1].argsort()[::-1]]\n    weights = np.where(labels[:,0]==0, 20, 1)\n    cut_vals = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n    gini = [0,0]\n    for i in [1,0]:\n        labels = np.transpose(np.array([y_true, y_pred]))\n        labels = labels[labels[:, i].argsort()[::-1]]\n        weight = np.where(labels[:,0]==0, 20, 1)\n        weight_random = np.cumsum(weight / np.sum(weight))\n        total_pos = np.sum(labels[:, 0] *  weight)\n        cum_pos_found = np.cumsum(labels[:, 0] * weight)\n        lorentz = cum_pos_found / total_pos\n        gini[i] = np.sum((lorentz - weight_random) * weight)\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.450203Z","iopub.status.idle":"2022-09-17T00:54:04.450645Z","shell.execute_reply.started":"2022-09-17T00:54:04.450414Z","shell.execute_reply":"2022-09-17T00:54:04.450456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Light GBM  \nWeighting the data and make some hyperparameter tuning","metadata":{}},{"cell_type":"code","source":"#from lightgbm import LGBMClassifier, plot_importance\n# import xgboost as xgb\n\n# Weighting imbalanced sample\nWEIGHTS = (y_tra.value_counts(normalize = True).min() / y_tra.value_counts(normalize = True))\nWEIGHTS = WEIGHTS.reset_index(drop=False)\nTRAIN_WEIGHTS = pd.DataFrame(y_tra.rename(columns ={'target':'old_target'})).merge(pd.DataFrame(WEIGHTS), how = 'left', left_on = 'old_target', right_on = 'target').iloc[:,-1]\nlgb_train = lgb.Dataset(X_tra, label=y_tra, weight = TRAIN_WEIGHTS)\n# lgb_train = lgb.Dataset(X_tra, label=y_tra)\nlgb_valid = lgb.Dataset(X_tes, y_tes)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.452448Z","iopub.status.idle":"2022-09-17T00:54:04.452861Z","shell.execute_reply.started":"2022-09-17T00:54:04.452665Z","shell.execute_reply":"2022-09-17T00:54:04.452685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cross validating model\n'''\ns = time.time()\n\nlgbm_params = {'learning_rate':0.03, \n              'boosting_type':'dart',   # dart, gbdt\n              'objective':'binary',\n              'seed': 42, \n              'metric':'binary_logloss',\n              'num_leaves':100,\n              'max_depth':10,\n              'feature_fraction': 0.40,\n              'bagging_freq': 10,\n              'bagging_fraction': 0.50,\n              'n_jobs': -1,\n              'lambda_l2': 2,\n              'min_data_in_leaf': 40, # larger to avoid overfitting, underfitting\n              }\nlgb_cv = lgb.cv(\n    lgbm_params, lgb_train, num_boost_round=2000, nfold=5, stratified=False, shuffle=True, metrics='binary_logloss',\n    early_stopping_rounds=50, verbose_eval=50, show_stdv=True, seed=0)\n\nprint('best n_estimators:', len(lgb_cv['binary_logloss-mean']))\nprint('best cv score:', lgb_cv['binary_logloss-mean'][-1])\n\ne = time.time()\ndeltat = e-s\nprint(deltat,\" second\")\n'''","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.454522Z","iopub.status.idle":"2022-09-17T00:54:04.455206Z","shell.execute_reply.started":"2022-09-17T00:54:04.455003Z","shell.execute_reply":"2022-09-17T00:54:04.455024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameter searching\n# kfold = StratifiedKFold(n_splits = 5)\n# param_grid = [{'solver': ['newton-cg', 'lbfgs','saga'],\n#               'max_iter':[100,300],\n#              'C':[0.05,0.5,1,50],\n#               'class_weight':[None,'balanced'], #{0: 1,1: 0.349}\n#               'penalty':['l1','l2']}]\n# model = LogisticRegression()\n# print(\"start searching...\")\n# grid = GridSearchCV(model, param_grid=param_grid, scoring='neg_log_loss', n_jobs=-1,cv=kfold,return_train_score=True,verbose = 10) #\n\n# grid.fit(pca_train,y_tra.values.ravel())\n# print('Best Penalty:', grid.best_estimator_.get_params()['penalty'])\n# print('Best C:', grid.best_estimator_.get_params()['C'])\n# print(); print(grid.best_estimator_.get_params())","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.456606Z","iopub.status.idle":"2022-09-17T00:54:04.457329Z","shell.execute_reply.started":"2022-09-17T00:54:04.457122Z","shell.execute_reply":"2022-09-17T00:54:04.457144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Finding optimal parameters\n# params_test1={\n#     'max_depth': range(3,10,50),\n#     'num_leaves':range(50, 100, 150)}\n\n# model_lgb = lgb.LGBMRegressor(objective='binary',\n#                               learning_rate=0.03, n_estimators=50, min_child_samples=40,\n#                               metric='binary_logloss', bagging_fraction = 0.5,feature_fraction = 0.4)\n# gsearch1 = GridSearchCV(estimator=model_lgb, param_grid=params_test1, scoring='neg_mean_squared_error', cv=5, verbose=1, n_jobs=4)\n# gsearch1.fit(X_tra, y_tra)\n# gsearch1.grid_scores_, gsearch1.best_params_, gsearch1.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.458969Z","iopub.status.idle":"2022-09-17T00:54:04.459670Z","shell.execute_reply.started":"2022-09-17T00:54:04.459453Z","shell.execute_reply":"2022-09-17T00:54:04.459480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# params_test2={\n#     'feature_fraction': [0.2, 0.4, 0.6, 0.8, 1.0],\n#     'bagging_fraction': [0.6, 0.7, 0.8, 0.9, 1.0]\n# }\n# model_lgb = lgb.LGBMRegressor(objective='binary',num_leaves=100,\n#                               learning_rate=0.03, n_estimators=50, max_depth=10, \n#                               metric='binary_logloss', bagging_freq = 10,  min_child_samples=40)\n# gsearch2 = GridSearchCV(estimator=model_lgb, param_grid=params_test2, scoring='neg_mean_squared_error', cv=5, verbose=1, n_jobs=4)\n# gsearch2.fit(X_tra, y_tra)\n# gsearch2.grid_scores_, gsearch2.best_params_, gsearch2.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.461749Z","iopub.status.idle":"2022-09-17T00:54:04.462290Z","shell.execute_reply.started":"2022-09-17T00:54:04.462067Z","shell.execute_reply":"2022-09-17T00:54:04.462092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# params_test3={\n#     'min_child_samples': [10, 20, 30, 40, 51],\n#     'min_child_weight':[0.001, 0.002]\n# }\n# model_lgb = lgb.LGBMRegressor(objective='binary',num_leaves=100,\n#                               learning_rate=0.03, n_estimators=50, max_depth=10, \n#                               metric='binary_logloss', bagging_fraction = 0.5, feature_fraction = 0.4)\n# gsearch3 = GridSearchCV(estimator=model_lgb, param_grid=params_test3, scoring='neg_mean_squared_error', cv=5, verbose=1, n_jobs=4)\n# gsearch3.fit(X_tra, y_tra)\n# gsearch3.grid_scores_, gsearch3.best_params_, gsearch3.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.464062Z","iopub.status.idle":"2022-09-17T00:54:04.464646Z","shell.execute_reply.started":"2022-09-17T00:54:04.464407Z","shell.execute_reply":"2022-09-17T00:54:04.464451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# params_test4={\n#     'boosting_type':['dart','gbdt']\n# }\n# model_lgb = lgb.LGBMRegressor(objective='binary',num_leaves=100,\n#                               learning_rate=0.03, n_estimators=50, max_depth=10, \n#                               metric='binary_logloss', bagging_fraction = 0.5,\n#                               feature_fraction = 0.4,min_child_samples=40)\n# gsearch4 = GridSearchCV(estimator=model_lgb, param_grid=params_test4, scoring='neg_mean_squared_error', cv=5, verbose=1, n_jobs=4)\n# gsearch4.fit(X_tra, y_tra)\n# gsearch4.grid_scores_, gsearch4.best_params_, gsearch4.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.465982Z","iopub.status.idle":"2022-09-17T00:54:04.466726Z","shell.execute_reply.started":"2022-09-17T00:54:04.466516Z","shell.execute_reply":"2022-09-17T00:54:04.466538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_params_best = {'learning_rate':0.03, \n              'boosting_type': 'dart',   # dart, gbdt gsearch4.best_params_['boosting_type']\n              'objective':'binary',\n              'seed': 42, \n              'metric':['binary_logloss','auc'],\n              'num_leaves':500,\n              'max_depth':20,\n              'feature_fraction': 0.40,\n              'bagging_freq': 10,\n              'bagging_fraction': 0.50,\n              'n_jobs': -1,\n              'lambda_l2': 2,\n              'min_data_in_leaf': 40, # larger to avoid overfitting, underfitting\n              }","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.468344Z","iopub.status.idle":"2022-09-17T00:54:04.468848Z","shell.execute_reply.started":"2022-09-17T00:54:04.468644Z","shell.execute_reply":"2022-09-17T00:54:04.468666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training optimal model\ns = time.time()\nevals_results = {}\nclf = lgb.train(lgbm_params_best,\n                lgb_train,\n                num_boost_round = 4000,  #2700\n                valid_sets = [lgb_train, lgb_valid],\n                verbose_eval = 10,\n                early_stopping_rounds = 50, \n                categorical_feature = cat_features,\n                evals_result = evals_results\n                ) \ne = time.time()\ndeltat = e-s\nprint(deltat,\" second\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.470835Z","iopub.status.idle":"2022-09-17T00:54:04.471580Z","shell.execute_reply.started":"2022-09-17T00:54:04.471333Z","shell.execute_reply":"2022-09-17T00:54:04.471354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Prediction on test(validating) data\ny_pred_lgbm=clf.predict(X_tes)\ny_pred_lgbm_class = y_pred_lgbm.copy()\nfor i in range(0, X_tes.shape[0]):\n    if y_pred_lgbm_class[i]>=.5:       \n        y_pred_lgbm_class[i]=1\n    else:\n        y_pred_lgbm_class[i]=0\n\n#Metrics\naccuracy_lgbm = metrics.accuracy_score(y_pred_lgbm_class,y_tes)\nprint (\"Accuracy with LGBM = \", accuracy_lgbm)\n\ncm_lgbm = confusion_matrix(y_tes, y_pred_lgbm_class)\nplt.figure(figsize=(9,9))\nsns.heatmap(cm_lgbm, annot=True, fmt=\"2d\", linewidths=.5, square = True, cmap = 'Blues');\nplt.ylabel('Actual label');\nplt.xlabel('Predicted label');\ntitle = 'LightGBM Accuracy Score: {:.4f}'.format(accuracy_lgbm)\nplt.title(title, size = 15);\nplt.savefig(\"../lgbm.png\",dpi=300)\nplt.savefig(\"lgbm CM.png\")\n\nroc_lgbm = roc_auc_score(y_tes,y_pred_lgbm_class)\nprint(\"AUC score with LGBM = \",roc_lgbm)\namex_metric_lgbm = amex_metric(y_tes.values.ravel(),y_pred_lgbm)\nprint(\"amex_metric with LGBM = \",amex_metric_lgbm)\namex_metric_lgbm2 = amex_metric2(y_tes.reset_index(drop=True),pd.DataFrame(y_pred_lgbm,columns= [\"prediction\"]))\nprint(\"amex_metric with LGBM2 = \",amex_metric_lgbm2)\nprint('precision_score :\\n',precision_score(y_tes,y_pred_lgbm_class,pos_label=1))\nprint('recall_score :\\n',recall_score(y_tes,y_pred_lgbm_class,pos_label=1))","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.473502Z","iopub.status.idle":"2022-09-17T00:54:04.473955Z","shell.execute_reply.started":"2022-09-17T00:54:04.473748Z","shell.execute_reply":"2022-09-17T00:54:04.473768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the confusion matrix, we can see that the false positive is much larger than false negative, which indicates a too conservative rating and overestimation.","metadata":{}},{"cell_type":"code","source":"# Save the model\nmodel_name = 'lgb' + datetime.datetime.now().strftime('%Y-%m-%d-%H_%M') + '.model'\nprint(model_name)\nclf.save_model(model_name)\n# clf.booster_.save_model(model_name)\n# joblib.dump(clf,'asame'+ model_name)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.475551Z","iopub.status.idle":"2022-09-17T00:54:04.476208Z","shell.execute_reply.started":"2022-09-17T00:54:04.476004Z","shell.execute_reply":"2022-09-17T00:54:04.476025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature Impoartance\nplt.rcParams[\"figure.figsize\"] = (9, 9)\nlgb.plot_importance(clf, max_num_features = 40, height=.9)\nplt.grid(False)\nplt.savefig(\"featIm.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.478230Z","iopub.status.idle":"2022-09-17T00:54:04.478753Z","shell.execute_reply.started":"2022-09-17T00:54:04.478499Z","shell.execute_reply":"2022-09-17T00:54:04.478521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ROC\nlgb.plot_metric(evals_results, metric = 'auc')\nplt.title(\"AUC curve of LightGBM\")\nplt.savefig(\"AUC Lgb.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.480278Z","iopub.status.idle":"2022-09-17T00:54:04.481012Z","shell.execute_reply.started":"2022-09-17T00:54:04.480786Z","shell.execute_reply":"2022-09-17T00:54:04.480814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# learning curve\nlgb.plot_metric(evals_results, metric = 'binary_logloss')\nplt.title(\"Learning curve of LightGBM\")\nplt.savefig(\"Lgb learning curve.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.483145Z","iopub.status.idle":"2022-09-17T00:54:04.483644Z","shell.execute_reply.started":"2022-09-17T00:54:04.483413Z","shell.execute_reply":"2022-09-17T00:54:04.483456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### PCA + Logistic Regression","metadata":{}},{"cell_type":"code","source":"# Decomposition\npca = PCA(n_components=100) # retain 95% variance\npca.fit(X_tra) #.iloc[0:200000,:]\npca_train = pca.transform(X_tra)\n\npca_test = pca.transform(X_tes)\n\n# if need to go back to original size\n# approximation = pca.inverse_transform(lower_dimensional_data)\npca.explained_variance_ratio_","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.485686Z","iopub.status.idle":"2022-09-17T00:54:04.486092Z","shell.execute_reply.started":"2022-09-17T00:54:04.485895Z","shell.execute_reply":"2022-09-17T00:54:04.485914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pcaDf_tra = pd.DataFrame(data = pca_train[:,:3]\n             , columns = ['principal component 1', 'principal component 2', 'principal component 3'])\npcaDf = pd.concat([pcaDf_tra.reset_index(drop=True),y_tra.reset_index(drop=True)],axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.487699Z","iopub.status.idle":"2022-09-17T00:54:04.488152Z","shell.execute_reply.started":"2022-09-17T00:54:04.487949Z","shell.execute_reply":"2022-09-17T00:54:04.487970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot the first 3 components in the 3-d plot\n# %matplotlib notebook\nfig = plt.figure(figsize=(9,9))\nax = fig.add_subplot(projection='3d')\nax.set_xlabel('Principal Component 1', fontsize = 15)\nax.set_ylabel('Principal Component 2', fontsize = 15)\nax.set_zlabel('Principal Component 3', fontsize = 15)\n\nax.set_title('3 components PCA', fontsize = 20)\ntargets = [0, 1]\ncolors = ['g', 'orange']\nmarks = ['o','o']\n# normalize = mpl.colors.Normalize(vmin=-1, vmax=1) # vmin=pcaDf.min().min(), vmax=pcaDf.max().max() , norm = normalize\nfor target, color, m in zip(targets,colors,marks):\n    indicesToKeep = pcaDf['target'] == target\n    ax.scatter(pcaDf.loc[indicesToKeep, 'principal component 1']\n               , pcaDf.loc[indicesToKeep, 'principal component 2']\n               , pcaDf.loc[indicesToKeep, 'principal component 3']   \n               , c = color\n               , marker=m\n               , label=target\n               , alpha = 0.3\n               , edgecolor = None\n               , s = 50,vmin=-10, vmax=30)\n# ax.scatter.legend_elements()\nax.legend(title = \"target\")\nax.grid(True)\nax.set_xlim(-3000,5000)\nax.set_ylim(-3000,4000)\nax.set_zlim(-3000,4000)\nax.view_init(60,20) \n# plt.show()\nplt.savefig(\"pca.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.490058Z","iopub.status.idle":"2022-09-17T00:54:04.490878Z","shell.execute_reply.started":"2022-09-17T00:54:04.490535Z","shell.execute_reply":"2022-09-17T00:54:04.490558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# logistic regression model\n# tdelta_list = []\n# solver = 'lbfgs'\ns = time.time()\nlogi = LogisticRegression(C = 0.05,\n                          solver = 'lbfgs',\n                          max_iter=1000,\n                          class_weight={0: 1,1: 0.349}) # lbfgs, newton-cg, saga, liblinear, grid.best_estimator_.get_params()\n# clf = GridSearchCV(model, parameters, cv = 10)\nlogi.fit(pca_train,y_tra.values.ravel())\ne = time.time()\ntdelta = e - s\nprint(tdelta, \"Seconds\")\n# tdelta_list.append({'time' : tdelta, 'bin' : i})\naccuracy_logi = logi.score(pca_test, y_tes)\ny_predict_logi = logi.predict_proba(pca_test)[:,1]\ny_predict_logi_class = logi.predict(pca_test)\nroc_logi = roc_auc_score(y_tes,y_predict_logi)\nprint(\"Accuracy with logistic regression = \",accuracy_logi)\nprint(\"roc score with logistic regression = \",roc_logi)\nprint('precision_score :\\n',precision_score(y_tes,y_predict_logi_class,pos_label=1))\nprint('recall_score :\\n',recall_score(y_tes,y_predict_logi_class,pos_label=1))\nmodel_name = 'logi' + datetime.datetime.now().strftime('%Y-%m-%d-%H_%M') + '.model'\njoblib.dump(logi,model_name)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.491915Z","iopub.status.idle":"2022-09-17T00:54:04.492331Z","shell.execute_reply.started":"2022-09-17T00:54:04.492128Z","shell.execute_reply":"2022-09-17T00:54:04.492147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric_logi = amex_metric(y_tes.values.ravel(),y_predict_logi) # pd.DataFrame(y_predict,columns= [\"prediction\"])\nprint(\"amex_metric with Logi = \",amex_metric_logi)\namex_metric_logi2 = amex_metric2(y_tes.reset_index(drop=True),pd.DataFrame(y_predict_logi,columns= [\"prediction\"]))\nprint(\"amex_metric with Logi2 = \",amex_metric_logi2)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.494387Z","iopub.status.idle":"2022-09-17T00:54:04.495158Z","shell.execute_reply.started":"2022-09-17T00:54:04.494885Z","shell.execute_reply":"2022-09-17T00:54:04.494907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# confusion matrix of LR\ncm_logi = confusion_matrix(y_predict_logi_class,y_tes)\nplt.figure(figsize=(9,9))\nsns.heatmap(cm_logi, annot=True, fmt=\"2d\", linewidths=.5, square = True, cmap = 'Blues');\nplt.ylabel('Actual label');\nplt.xlabel('Predicted label');\ntitle = 'Logistic Regression Accuracy Score: {:.4f}'.format(accuracy_logi)\nplt.title(title, size = 15);\n# plt.show()\nplt.savefig(\"logi CM.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.497301Z","iopub.status.idle":"2022-09-17T00:54:04.497961Z","shell.execute_reply.started":"2022-09-17T00:54:04.497742Z","shell.execute_reply":"2022-09-17T00:54:04.497765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression with SGD","metadata":{}},{"cell_type":"code","source":"# Logistic Regression with SGD solver and PCA\nfrom sklearn.linear_model import SGDClassifier\n# from sklearn.calibration import CalibratedClassifierCV\ns = time.time()\nsgd = SGDClassifier(loss='log',\n                  penalty='elasticnet',\n                  max_iter=1000,\n                  n_jobs=-1)\n# sgd = CalibratedClassifierCV(base_model)\n\nsgd.fit(pca_train,y_tra.values.ravel())\ne = time.time()\ntdelta = e - s\nprint(tdelta)\naccuracy_sgd = sgd.score(pca_test, y_tes)\ny_predict_sgd = sgd.predict(pca_test)\n# y_predict_sgd_class = sgd.predict(pca_test)\n\nroc_sgd = roc_auc_score(y_tes,y_predict_sgd)\nprint(\"Accuracy with SGD logistic regression = \",accuracy_sgd)\nprint(\"roc score with SGD logistic regression = \",roc_sgd)\nprint('precision_score :\\n',precision_score(y_tes,y_predict_sgd,pos_label=1))\nprint('recall_score :\\n',recall_score(y_tes,y_predict_sgd,pos_label=1))\namex_metric_sgd = amex_metric(y_tes.values.ravel(),y_predict_sgd)\nprint(\"amex_metric with SGD = \",amex_metric_sgd)\namex_metric_sgd2 = amex_metric2(y_tes.reset_index(drop=True),pd.DataFrame(y_predict_sgd,columns= [\"prediction\"]))\nprint(\"amex_metric with SGD = \",amex_metric_sgd2)\n\n# save model\n# model_name = 'sgd' + datetime.datetime.now().strftime('%Y-%m-%d-%H_%M') + '.model'\n# joblib.dump(sgd, model_name)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.499371Z","iopub.status.idle":"2022-09-17T00:54:04.500131Z","shell.execute_reply.started":"2022-09-17T00:54:04.499921Z","shell.execute_reply":"2022-09-17T00:54:04.499942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_scores = cross_val_predict(classifier, x_train, y_train, cv=3,\n#                              method=\"decision_function\")\n# precisions, recalls, thresholds = precision_recall_curve(y_train, y_scores)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-09-17T00:54:04.501647Z","iopub.status.idle":"2022-09-17T00:54:04.502332Z","shell.execute_reply.started":"2022-09-17T00:54:04.502115Z","shell.execute_reply":"2022-09-17T00:54:04.502137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Decision Tree","metadata":{}},{"cell_type":"code","source":"# DT training\nDT = DecisionTreeClassifier()\n\n# DT_param = {'max_depth': np.linspace(10, 50, 5, endpoint=True),\n#             'max_features': np.arange(1,len(features),10),\n# #             'min_samples_leaf': np.linspace(0.1, 0.5, 20, endpoint=True),\n# #             'min_samples_split': np.linspace(0.001, 0.1,10 endpoint=True),\n#             'criterion': ['gini', 'entropy']}\n\n# DT_cv = RandomizedSearchCV(DT, param_distributions=DT_param, cv=5, n_iter=20,n_jobs=-1, random_state=0)\n# DT_cv.fit(X_tra, y_tra)\n# print(DT_cv.best_params_)\n\nDT_best = DecisionTreeClassifier(min_samples_leaf = 2,\n                                 min_samples_split = 10,\n                                 class_weight = {0: 1,1: 0.349},\n                                 max_features = 170,\n                                 max_depth = 50,\n                                 criterion = 'gini',\n                                 random_state = 0)\n\ns = time.time()\nDT_best.fit(X_tra, y_tra)\ne = time.time()\ntdelta = e - s\nprint(tdelta, \"Seconds\")\n# save model\n# model_name = 'DT' + datetime.datetime.now().strftime('%Y-%m-%d-%H_%M') + '.model'\n# joblib.dump(DT_best, model_name)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.504406Z","iopub.status.idle":"2022-09-17T00:54:04.505406Z","shell.execute_reply.started":"2022-09-17T00:54:04.505173Z","shell.execute_reply":"2022-09-17T00:54:04.505201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cv_scores = cross_validate(DT_best, X_tra, y_tra, scoring='accuracy', cv=5)\n# score_mean = round(cv_scores['test_score'].mean(), 5)\n# score_std = round(cv_scores['test_score'].std(), 5)\n# print(f'CV Score of Decision Tree: {score_mean} ({score_std})')","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.507257Z","iopub.status.idle":"2022-09-17T00:54:04.507679Z","shell.execute_reply.started":"2022-09-17T00:54:04.507473Z","shell.execute_reply":"2022-09-17T00:54:04.507500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Prediction on test data\ny_pred_DT = DT_best.predict_proba(X_tes)[:,1]\ny_predict_DT_class = DT_best.predict(X_tes)\n\n# for i in range(0, X_tes.shape[0]):\n#     if y_pred_DT[i] >= .5:\n#         y_pred_DT[i] = 1\n#     else:\n#         y_pred_DT[i] = 0","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.508951Z","iopub.status.idle":"2022-09-17T00:54:04.509382Z","shell.execute_reply.started":"2022-09-17T00:54:04.509147Z","shell.execute_reply":"2022-09-17T00:54:04.509166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Metrics\naccuracy_DT = metrics.accuracy_score(y_predict_DT_class,y_tes)\nprint (\"Accuracy with Decision Tree = \", accuracy_DT)\n\ncm_DT = confusion_matrix(y_tes, y_predict_DT_class)\nplt.figure(figsize=(9,9))\nsns.heatmap(cm_DT, annot=True, fmt=\"2d\", linewidths=.5, square = True, cmap = 'Blues');\nplt.ylabel('Actual label');\nplt.xlabel('Predicted label');\ntitle = 'Decision Tree Accuracy Score: {:.4f}'.format(accuracy_DT)\nplt.title(title, size = 15);\n# plt.show()\nplt.savefig(\"DT CM.png\")\nroc_DT = roc_auc_score(y_tes,y_predict_DT_class)\nprint(\"AUC score with Decision Tree = \",roc_DT)\namex_metric_DT = amex_metric(y_tes.values.ravel(),y_pred_DT)\nprint(\"amex_metric with Decision Tree = \",amex_metric_DT)\namex_metric_DT2 = amex_metric2(y_tes.reset_index(drop=True),pd.DataFrame(y_pred_DT,columns= [\"prediction\"]))\nprint(\"amex_metric with Decision Tree = \",amex_metric_DT2)\nprint('precision_score :\\n',precision_score(y_tes,y_predict_DT_class,pos_label=1))\nprint('recall_score :\\n',recall_score(y_tes,y_predict_DT_class,pos_label=1))","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.511276Z","iopub.status.idle":"2022-09-17T00:54:04.512114Z","shell.execute_reply.started":"2022-09-17T00:54:04.511903Z","shell.execute_reply":"2022-09-17T00:54:04.511926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Random Forest","metadata":{}},{"cell_type":"code","source":"# Random Forest\nfrom sklearn.ensemble import RandomForestClassifier\nrf = RandomForestClassifier(\n            warm_start=True,\n            max_features=\"log2\",\n            oob_score=True,\n            random_state=42,\n        )\n\ns = time.time()\nrf.fit(X_tra, y_tra.values.ravel())\ne = time.time()\ntdelta = e - s\nprint(tdelta)\n\n# save model\nmodel_name = 'RF' + datetime.datetime.now().strftime('%Y-%m-%d-%H_%M') + '.model'\njoblib.dump(rf, model_name)\n\n#Prediction on test data\ny_pred_rf=rf.predict_proba(X_tes)[:,1]\n# for i in range(0, X_tes.shape[0]):\n#     if y_pred_rf[i]>=.5:      \n#         y_pred_rf[i]=1\n#     else:\n#         y_pred_rf[i]=0\ny_predict_rf_class = rf.predict(X_tes)\n\n# Metrics\naccuracy_rf = metrics.accuracy_score(y_predict_rf_class,y_tes)\nroc_rf = roc_auc_score(y_tes,y_predict_rf_class)\nprint(\"Accuracy with Random Forest = \",accuracy_rf)\nprint(\"roc score with Random Forest = \",roc_rf)\nprint('precision_score :\\n',precision_score(y_tes,y_predict_rf_class,pos_label=1))\nprint('recall_score :\\n',recall_score(y_tes,y_predict_rf_class,pos_label=1))\namex_metric_rf = amex_metric(y_tes.values.ravel(),y_pred_rf)\nprint(\"amex_metric with RF = \",amex_metric_rf)\namex_metric_rf2 = amex_metric2(y_tes.reset_index(drop=True),pd.DataFrame(y_pred_rf,columns= [\"prediction\"]))\nprint(\"amex_metric with RF = \",amex_metric_rf2)","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.513211Z","iopub.status.idle":"2022-09-17T00:54:04.513623Z","shell.execute_reply.started":"2022-09-17T00:54:04.513399Z","shell.execute_reply":"2022-09-17T00:54:04.513416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# confusion matrix \ncm_RF = confusion_matrix(y_tes, y_predict_rf_class)\nplt.figure(figsize=(9,9))\nsns.heatmap(cm_RF, annot=True, fmt=\"2d\", linewidths=.5, square = True, cmap = 'Blues');\nplt.ylabel('Actual label');\nplt.xlabel('Predicted label');\ntitle = 'Random Forest Accuracy Score: {:.4f}'.format(accuracy_rf)\nplt.title(title, size = 15);\n# plt.show()\nplt.savefig(\"RF CM.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.514736Z","iopub.status.idle":"2022-09-17T00:54:04.515166Z","shell.execute_reply.started":"2022-09-17T00:54:04.514977Z","shell.execute_reply":"2022-09-17T00:54:04.514996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The strength of DT/random forest is they don't suffer from unbalanced dataset","metadata":{}},{"cell_type":"markdown","source":"### HistGradientBoosting","metadata":{}},{"cell_type":"code","source":"# Gradient Boosting Algorithm\n\nhgb = HistGradientBoostingClassifier(\n                 max_iter = 700,  #2700\n                 verbose = 2,\n                 n_iter_no_change = 50,\n                 learning_rate = 0.03,\n                 loss = 'binary_crossentropy',\n                 max_leaf_nodes = 100,\n                 min_samples_leaf =  40, # larger to avoid overfitting, underfitting\n                 max_depth = 20,\n                 categorical_features = cat_features)\ns = time.time()\n\nhgb.fit(X_tra, y_tra.values.ravel())\n\ne = time.time()\ntdelta = e - s\n\n# hgb.score(X_tes, y_tes)\n\nprint(tdelta)\n# save model\nmodel_name = 'hgb' + datetime.datetime.now().strftime('%Y-%m-%d-%H_%M') + '.model'\njoblib.dump(hgb, model_name)\n\n#Prediction on test data\ny_pred_hgb=hgb.predict_proba(X_tes)[:,1]  # predict_proba(X)\n# for i in range(0, X_tes.shape[0]):\n#     if y_pred_hgb[i]>=.5:      \n#         y_pred_hgb[i]=1\n#     else:\n#         y_pred_hgb[i]=0\ny_predict_hgb_class = hgb.predict(X_tes)\n        \n# Metrics\naccuracy_hgb = metrics.accuracy_score(y_predict_hgb_class,y_tes)\nroc_hgb = roc_auc_score(y_tes,y_predict_hgb_class)\nprint(\"Accuracy with HistGradientBoostingClassifier = \",accuracy_hgb)\nprint(\"roc score with HistGradientBoostingClassifier = \",roc_hgb)\nprint('precision_score :\\n',precision_score(y_tes,y_predict_hgb_class,pos_label=1))\nprint('recall_score :\\n',recall_score(y_tes,y_predict_hgb_class,pos_label=1))\namex_metric_hgb = amex_metric(y_tes.values.ravel(),y_pred_hgb)\nprint(\"amex_metric with hgb = \",amex_metric_hgb)\namex_metric_hgb2 = amex_metric2(y_tes.reset_index(drop=True),pd.DataFrame(y_pred_hgb,columns= [\"prediction\"]))\nprint(\"amex_metric with hgb = \",amex_metric_hgb2)\n\ncm_hgb = confusion_matrix(y_tes, y_predict_hgb_class)\nplt.figure(figsize=(9,9))\nsns.heatmap(cm_hgb, annot=True, fmt=\"2d\", linewidths=.5, square = True, cmap = 'Blues');\nplt.ylabel('Actual label');\nplt.xlabel('Predicted label');\ntitle = 'Hist Gradient Boosting Accuracy Score: {:.4f}'.format(accuracy_hgb)\nplt.title(title, size = 15);\nplt.savefig(\"hgb CM.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.517265Z","iopub.status.idle":"2022-09-17T00:54:04.517694Z","shell.execute_reply.started":"2022-09-17T00:54:04.517490Z","shell.execute_reply":"2022-09-17T00:54:04.517510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Comparison of results","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import RocCurveDisplay\n\n# plt.figure(figsize=(9,18))\n\nfig, (ax1, ax2) = plt.subplots(1,2,figsize= (18,9))\n\nmodels = [\n#     (\"LR wiht PCA\", logi),\n    (\"RF\", rf),\n#     (\"SGD LR with PCA\", sgd),\n    (\"DT\", DT_best),\n#     (\"KNN\", knn),\n#     (\"KNN w/o PCA\", knn2),\n    (\"Hist Gradient Boosting\", hgb)\n#     (\"LightGBM\", clf)\n]\n\nmodel_displays = {}\nfor name, pipeline in models:\n    model_displays[name] = RocCurveDisplay.from_estimator(\n        pipeline, X_tes, y_tes, ax=ax1, name=name\n    )\n\nmodels2 = [\n    (\"LR wiht PCA\", logi),\n#     (\"RF\", rf),\n    (\"SGD LR with PCA\", sgd),\n#     (\"DT\", DT),\n#     (\"KNN\", knn),\n#     (\"KNN w/o PCA\", knn2),\n#     (\"Hist Gradient Boosting\", hgb),\n#     (\"LightGBM\", lgb)\n]\n\nmodel_displays2 = {}\nfor name, pipeline in models2:\n    model_displays[name] = RocCurveDisplay.from_estimator(\n        pipeline, pca_test, y_tes, ax=ax1, name=name\n    )\n_ = ax1.set_title(\"ROC curve\")    \n\nlgb.plot_metric(evals_results, ax = ax2, metric = 'auc') ## could plot with roc_auc_curve\nax2.set_title(f\"AUC of LightGBM = {roc_lgbm:.3f}\")\nplt.savefig(\"ROC Curves.png\")","metadata":{"execution":{"iopub.status.busy":"2022-09-17T00:54:04.518894Z","iopub.status.idle":"2022-09-17T00:54:04.519322Z","shell.execute_reply.started":"2022-09-17T00:54:04.519087Z","shell.execute_reply":"2022-09-17T00:54:04.519104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Output done in another part -- Model test","metadata":{}}]}