{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install xgboost==0.90 -q\n!pip install pytorch_tabnet -q\n!pip install catboost -q","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-23T11:47:16.626008Z","iopub.execute_input":"2022-08-23T11:47:16.626652Z","iopub.status.idle":"2022-08-23T11:47:48.217134Z","shell.execute_reply.started":"2022-08-23T11:47:16.626509Z","shell.execute_reply":"2022-08-23T11:47:48.215422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport pandas as pd\nimport shutil\nimport random\n#Models\n#Logistic Regression\n\nfrom scipy.stats import t\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nfrom sklearn.experimental import enable_halving_search_cv \n#Dicision Tree Classifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import RandomizedSearchCV, HalvingRandomSearchCV\n#Ensenble\nfrom sklearn.ensemble import VotingClassifier,RandomForestClassifier\nfrom  sklearn import preprocessing\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, accuracy_score,roc_auc_score,f1_score,precision_score,recall_score, make_scorer\nfrom sklearn.model_selection import KFold,StratifiedKFold\nimport copy \nfrom sklearn.preprocessing import PowerTransformer, QuantileTransformer\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom scipy.stats import uniform\nimport os\n\nsns.set_style(\"darkgrid\", {\"grid.color\": \".6\", \"grid.linestyle\": \":\"})\n\nimport torch\n\nimport warnings\nfrom sklearn.exceptions import DataConversionWarning\nwarnings.filterwarnings(action='ignore', category=DataConversionWarning)\n\nfrom sklearn.decomposition import PCA\n\nimport os\nimport psutil\nimport gc\nimport itertools\n\nimport cudf\nimport dask_cudf\nimport cuml\n\n\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.tab_model import TabNetClassifier\n\nfrom tqdm.auto import tqdm\nimport joblib\nimport re\nfrom sklearn.preprocessing import PowerTransformer, QuantileTransformer\nfrom catboost import CatBoostClassifier\nfrom pytorch_tabnet.metrics import Metric\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.decomposition import PCA\n\ndef fix_name(x):\n    return x.split('_f')[0] +'_'+ (x.split('_f')[1].split('_')[1])\n\n#Color fonts\nclass bcolors:\n    HEADER = '\\033[95m'\n    OKBLUE = '\\033[94m'\n    OKCYAN = '\\033[96m'\n    OKGREEN = '\\033[92m'\n    WARNING = '\\033[93m'\n    FAIL = '\\033[91m'\n    ENDC = '\\033[0m'\n    BOLD = '\\033[1m'\n    UNDERLINE = '\\033[4m'\n    YELLOW = '\\033[93m'\n    pink = '\\033[95m'\n\n    def t_o_f(value):\n        if value:\n            return f'{bcolors.OKGREEN+str(value)+bcolors.ENDC}'\n        return f'{bcolors.FAIL+str(value)+bcolors.ENDC}'\n\n    def yes_no(yes, no):\n        return f'{bcolors.OKGREEN+str(yes)+bcolors.ENDC}', f'{bcolors.FAIL+str(no)+bcolors.ENDC}'\n\n    def yn_str(string,yes, no):\n        yes_2 = f'{bcolors.OKGREEN+str(yes)+bcolors.ENDC}'\n        no_2 =  f'{bcolors.FAIL+str(no)+bcolors.ENDC}'\n\n        string = re.sub(r\"\\b%s\\b\" % yes, yes_2, string)\n        string = re.sub(r\"\\b%s\\b\" % no, no_2, string)\n\n        return string","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-23T11:47:48.220971Z","iopub.execute_input":"2022-08-23T11:47:48.221518Z","iopub.status.idle":"2022-08-23T11:47:57.377305Z","shell.execute_reply.started":"2022-08-23T11:47:48.221461Z","shell.execute_reply":"2022-08-23T11:47:57.375951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"01\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Introduction </p>\n<div style=\"font-family: courier; font-size:16px\">\n    \n<li> This notebook is my, almost, complete pipeline created for the AMEX Competion. With it, I achieve 0.794 with a \"COMPLETE_MODEL\".\n<li> In this pipeline, is possible to train different models (CATBOOST | XGB | TABNET | LOGIT | LGBM) with Crossvalidation and different set of variables ( see section Models variables) that are constructed in the feature enginnering section. I use the treated version of the AMEX dataset shared by user RADDAR (<a url = \"https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format?select=test.parquet\">Link</a>) as base.\n<li> Also, I used an ensemble method with Genetic Algorithm that I shared in a previous post (<a url = \"https://www.kaggle.com/code/paulojunqueira/ensemble-optimization-with-ga\">Link<\\a>). \n<li> Needs GPU on and some models variables set may not work in kaggle kernel\n</div>","metadata":{}},{"cell_type":"markdown","source":"# Functions","metadata":{}},{"cell_type":"markdown","source":"## Models Selections ( variables)","metadata":{}},{"cell_type":"code","source":"def models_placeholder(cols, verbose = True, created = True):\n    vars_groups = {'S_MODEL':[],\"D_MODEL\":[],\n            \"B_MODEL\":[],\"R_MODEL\":[],\"P_MODEL\":[]}\n\n\n    custom_model = ['D_48', 'B_9', 'D_61', 'D_44', 'D_55', 'D_75', 'B_3', 'D_58', 'B_7', 'B_23', 'D_47', 'D_112', 'P_3', 'D_62', 'D_77', 'B_33', 'B_18', 'B_2', 'P_2', 'D_87']\n    importance_model =['P_2_mean', 'P_2_min', 'P_2_last', 'B_18_last', 'B_9_last', 'D_48_last',\n       'R_2_last', 'D_44_last', 'B_1_last', 'B_22_last', 'B_33_last',\n       'P_2_max', 'R_2_max', 'B_2_last', 'R_1_last', 'R_1_std', 'R_1_mean',\n       'R_4_last', 'B_11_last', 'R_5_last', 'B_37_last', 'R_1_max', 'B_1_mean',\n       'D_48_mean', 'D_41_last', 'D_42_min', 'D_44_max', 'D_42_mean',\n       'B_8_min', 'D_65_last']\n\n    for c in cols:\n        if c not in ['customer_ID','S_2','customer_ID_','customer_ID_2']:\n            if 'customer_ID' == c:\n                pass\n            elif (\"S\" in c):\n                vars_groups['S_MODEL'].append(c)\n            elif (\"D\" in c) :\n                vars_groups['D_MODEL'].append(c)\n            elif (\"B\" in c):\n                vars_groups['B_MODEL'].append(c)\n            elif (\"R\" in c):\n                vars_groups['R_MODEL'].append(c)\n            elif (\"P\" in c):\n                vars_groups['P_MODEL'].append(c)\n            elif (\"C\" in c):\n                for k in vars_groups.keys():\n                    vars_groups[k].append(c)\n\n    super = list(vars_groups.values())[0].copy()\n    for c in list(vars_groups.values())[1:]:\n        super +=c\n\n    vars_groups['COMPLETE_MODEL'] = list(set(super))\n\n    # vars_groups['CUSTOM_MODEL'] = vars_groups['D_MODEL'] + ['P_2', 'S_3', 'B_3', 'B_4']\n\n    model_combination = [i[0]+'-'+i[1]+'_MODEL' for i in list(itertools.combinations(['S','D','B','R','P'],2)) if i != 'COMPLETE_MODEL']\n    \n    #Creating combination groups\n    for i in model_combination:\n        a,b = i.split('_')[0].split('-')[0:2]\n        vars_groups[i] = list(set(vars_groups[a+'_MODEL'] + vars_groups[b+'_MODEL']))\n\n\n    model_combination = [i[0]+'-'+i[1]+'-'+i[2]+'_MODEL' for i in list(itertools.combinations(['S','D','B','R','P'],3)) if i != 'COMPLETE_MODEL']\n    \n    #Creating combination groups\n    for i in model_combination:\n        a,b,c= i.split('_')[0].split('-')[0:3]\n        vars_groups[i] = list(set(vars_groups[a+'_MODEL'] + vars_groups[b+'_MODEL'] + vars_groups[c+'_MODEL']))\n    \n\n\n    vars_str = '  '.join(list(cols))\n    m_v = []\n    for c in custom_model:\n        c = (c+'_')\n        st ='%s\\\\w+' % c\n        x = re.findall(st, vars_str)\n        m_v.append(x)\n    vars_groups['CUSTOM_MODEL'] = [j for i in m_v for j in i]\n\n    vars_groups['IMPORTANCE_MODEL'] = importance_model\n\n    vars_groups['LAST_MODEL'] = [i for i in vars_groups['COMPLETE_MODEL'] if i.endswith('_last')]\n    vars_groups['MEAN_MODEL'] = [i for i in vars_groups['COMPLETE_MODEL'] if i.endswith('_mean')]\n\n    \n\n    if verbose:\n        print(' Models:', vars_groups.keys(),'\\n','Number of Models:', len(vars_groups))\n    return vars_groups","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:47:57.380790Z","iopub.execute_input":"2022-08-23T11:47:57.381992Z","iopub.status.idle":"2022-08-23T11:47:57.405062Z","shell.execute_reply.started":"2022-08-23T11:47:57.381945Z","shell.execute_reply":"2022-08-23T11:47:57.403285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Features Engineering","metadata":{}},{"cell_type":"code","source":"def time_variables(df):\n    df['S_2'] =  cudf.to_datetime(df['S_2'])\n    df['C_week'] = df['S_2'].dt.isocalendar().week\n    df['C_month'] = df['S_2'].dt.month\n    df['C_day'] = df['S_2'].dt.day\n\n    return df\n\n\ndef create_var_PCA(df):\n    pca = PCA(n_components = 0.96 )\n    v_PCA = pca.fit_transform(df.drop('customer_ID_', axis =1 ).to_pandas().values)\n    columns_PCA = [f'C_PCA{i+1}'for i in range(v_PCA.shape[1])]\n    PCA_df = cudf.DataFrame(v_PCA, columns = columns_PCA)\n    # PCA_df['customer_ID_'] = df_sg.customer_ID_\n    # PCA_df['target'] = df_sg.target\n\n    df[columns_PCA] = PCA_df\n    del PCA_df\n\n    return df\n\n#lag features\ndef diff_variables(df,a = 'last',b = 'mean', operator = '-'):\n    if operator == '-':\n        for last_var, mean_var in zip([i for i in df.columns if i.endswith(a)],[i for i in df.columns if i.endswith(b)]):\n            name = \"_\".join(last_var.split(\"_\", 2)[:2])\n            df[f'{name}_{a}-{b}'] = df[last_var] - df[mean_var]\n    elif operator == '/':\n        for last_var, mean_var in zip([i for i in df.columns if i.endswith(a)],[i for i in df.columns if i.endswith(b)]):\n            name = \"_\".join(last_var.split(\"_\", 2)[:2])\n            df[f'{name}_{a}/{b}'] = df[last_var] / df[mean_var]\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:47:57.410054Z","iopub.execute_input":"2022-08-23T11:47:57.413354Z","iopub.status.idle":"2022-08-23T11:47:57.428127Z","shell.execute_reply.started":"2022-08-23T11:47:57.413310Z","shell.execute_reply":"2022-08-23T11:47:57.426940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from itertools import chain, combinations\n # Dont  use it\n\ndef combinations_feat(df):\n    combs = list(combinations([i for i in df.columns if i not in ['customer_ID', 'customer_ID_', 'target', 'S_2' ]], 2))\n\n    # combs = [i[0] + '_' + i[1] for i in list(combinations([i for i in df.columns if i not in ['customer_ID', 'customer_ID_', 'target', 'S_2' ]], 2))]\n\n    for a,b in combs:\n        # df[a + '+' + b] = df[a] + df[b]\n        # df[a + '/' + b] = df[a] / df[b]\n        df[a + '*' + b] = df[a] *  df[b]\n        df[a + '-' + b] = df[a] - df[b]\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:47:57.432728Z","iopub.execute_input":"2022-08-23T11:47:57.433192Z","iopub.status.idle":"2022-08-23T11:47:57.448174Z","shell.execute_reply.started":"2022-08-23T11:47:57.433161Z","shell.execute_reply":"2022-08-23T11:47:57.446819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Main\ndef features_engineering(df,model_name=None, train = False, type_agg = 2, vars_to_drop = '', change_data = False,  transform = False, fill_num = 0, cat_features = '', features_model = ''):\n    \n   ### Created Variables ----------------------------------------------------\n\n    # Time - not wotking in kaggle \n#     df = time_variables(df)\n\n\n    # #aggregations list \n    aggNumLi = ['first','mean', 'max','min','std', 'last'] \n    aggCatLi = ['count', 'nunique', 'last'] \n    \n    # aggNumLi = ['last'] \n    # aggCatLi = ['last'] \n    \n    cat_features = [i for i in cat_features if i  not in vars_to_drop]\n    \n    cols = df.columns\n    num_features = [i for i in cols if i not in ['customer_ID', 'S_2']+cat_features]\n    # a Quantile linear transformation\n    transform_ = QuantileTransformer()\n    # transform_ = PowerTransformer()\n\n   \n    ### Fillnas --------------------------------------------------------------\n    if change_data:\n        vars_type ={}\n        for i in cols:\n            vars_type[i] =df[i].dtype\n\n        for i in cols:\n            if i not in ['customer_ID', 'S_2']+cat_features:\n                df[i] = df[i].fillna(fill_num).astype(vars_type[i])\n                if transform:\n                    df[i] = cudf.Series(transform_.fit_transform(df[i].to_numpy().reshape(-1, 1))[:,0])\n\n            \n            elif i in cat_features:\n                t = vars_type[i]\n                if t == np.float32:\n                    # a = -1.0\n                    a = fill_num\n                    t = np.int32\n                elif t == 'O':\n                    a = str(fill_num)\n                else:\n                    a = fill_num\n                df[i] = df[i].astype(t).fillna(a)\n\n        df['S_2'] =  cudf.to_datetime(df['S_2'])\n    # df[num_features] = transform_qt.fit_transform(df[num_features].to_numpy())\n\n    ### Agregation --------------------------------------------------------------\n    if type_agg == 1:\n        df_sg = df.groupby('customer_ID').last() \n\n    elif type_agg == 2:\n        aggVars = {i:aggNumLi for i in cols if i not in ['target', 'customer_ID', 'S_2'] + cat_features}\n        if train:\n            aggVars['target'] = ['last']\n        aggCatVars = {i:aggCatLi for i in cat_features}\n        aggVars = {**aggVars, **aggCatVars}\n        df_sg = df.sort_values('S_2').groupby('customer_ID').agg(aggVars).reset_index()\n        df_sg.columns = [\"_\".join(pair) for pair in df_sg.columns]\n        if train:\n            df_sg = df_sg.rename(columns={\"target_last\": \"target\"})\n    del df\n            \n    if True:        \n        df_sg = df_sg.fillna(fill_num)\n\n    df_sg = diff_variables(df_sg)\n    df_sg = diff_variables(df_sg, 'first', 'last', operator='-')\n    df_sg = diff_variables(df_sg, 'first', 'last', operator='/')\n\n\n    vars = models_placeholder(df_sg.columns, False)\n    cat_features_agg = [i+f'_{h}'for i, j in aggCatVars.items() for h in j]\n\n    # df_sg = combinations_feat(df_sg)\n\n    # df_sg = create_var_PCA(df_sg)\n\n    df_sg = df_sg.round(2)\n    \n\n    if model_name:\n        vars = vars[model_name]\n        if features_model:\n            return df_sg[features_model]\n        else:\n            return df_sg[vars]\n    else:\n        return df_sg, vars, cat_features_agg\n\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:47:57.451829Z","iopub.execute_input":"2022-08-23T11:47:57.452378Z","iopub.status.idle":"2022-08-23T11:47:57.477075Z","shell.execute_reply.started":"2022-08-23T11:47:57.452289Z","shell.execute_reply":"2022-08-23T11:47:57.475044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Balacing ","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import resample\n\ndef balance(df_sg, return_concat = True):\n\n    df_down = resample(df_sg[df_sg['target'] == 0],\n                replace=True,\n                n_samples=len(df_sg[df_sg['target'] == 1]),\n                random_state=42)\n    if return_concat:\n        return cudf.concat([df_down, df_sg[df_sg['target'] == 1]])\n\n    return df_down","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:47:57.479860Z","iopub.execute_input":"2022-08-23T11:47:57.480563Z","iopub.status.idle":"2022-08-23T11:47:57.496187Z","shell.execute_reply.started":"2022-08-23T11:47:57.480521Z","shell.execute_reply":"2022-08-23T11:47:57.494590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Categorical and to Drop variables","metadata":{}},{"cell_type":"code","source":"# cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\",'C_week','C_month','C_day']\ncat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n\nto_remove = []","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:47:57.497928Z","iopub.execute_input":"2022-08-23T11:47:57.501299Z","iopub.status.idle":"2022-08-23T11:47:57.510135Z","shell.execute_reply.started":"2022-08-23T11:47:57.501256Z","shell.execute_reply":"2022-08-23T11:47:57.508720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load and Aggregate Dataset","metadata":{}},{"cell_type":"code","source":"# path_train = '/content/drive/MyDrive/Kaggle/AMEX_2022/Output/Train'\npath_train = '../input/amex-data-integer-dtypes-parquet-format/'\npath_labels = '../input/amex-default-prediction/train_labels.csv'\n\ntrain_files = [i for i in os.listdir(path_train) if i.startswith('train')]\nprint(train_files)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:47:57.512499Z","iopub.execute_input":"2022-08-23T11:47:57.513617Z","iopub.status.idle":"2022-08-23T11:47:57.528846Z","shell.execute_reply.started":"2022-08-23T11:47:57.513565Z","shell.execute_reply":"2022-08-23T11:47:57.526891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"df_sg = cudf.DataFrame()\ny = []\nagg = True\ndown = False\n\nfor ni, block in enumerate(train_files):\n\n    print(f'Block {ni+1}/{len(train_files)}')\n    df = cudf.read_parquet((path_train+'/'+block)).drop(['customer_ID_2']+to_remove, axis = 1, errors='ignore')\n    label_df = cudf.read_csv(path_labels)\n    df = df.merge( label_df, on = 'customer_ID')\n    \n    if agg:\n        df, vars_groups, cat_features_agg = features_engineering(df, train = True, vars_to_drop = to_remove, transform = False, fill_num = -127, cat_features = cat_features)\n    y.append(df['target'].to_pandas().values)\n    df_sg = cudf.concat([df_sg, df])\n    del df\n\n\nif down:\n    print('Downsampling Class 0 ...')\n    df_sg = balance(df_sg, True)\n    label_df = df_sg[['customer_ID_', 'target']]\n    y = df_sg['target'].to_pandas()\nelse:\n    y = pd.Series(np.concatenate(y))\nprint('Total of columns available:', len(df_sg.columns))","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:47:57.535858Z","iopub.execute_input":"2022-08-23T11:47:57.537437Z","iopub.status.idle":"2022-08-23T11:48:43.860830Z","shell.execute_reply.started":"2022-08-23T11:47:57.537392Z","shell.execute_reply":"2022-08-23T11:48:43.859280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Models variables\n- Models Variables to select in traning phase","metadata":{}},{"cell_type":"code","source":"#Check DF\nvars_groups = models_placeholder(df_sg.columns)\nprint('Number of variables for each model ')\nfor model, i in vars_groups.items():\n    print(model, len(i), ' variables')","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:48:43.863163Z","iopub.execute_input":"2022-08-23T11:48:43.863649Z","iopub.status.idle":"2022-08-23T11:48:43.876590Z","shell.execute_reply.started":"2022-08-23T11:48:43.863593Z","shell.execute_reply":"2022-08-23T11:48:43.874950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metrics","metadata":{}},{"cell_type":"code","source":"def amex_metric( y_true, y_pred ):\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame):\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame):\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) :\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n    \n    \n    y_pred = pd.DataFrame(y_pred, columns =['prediction'])\n    y_true = pd.DataFrame(y_true, columns =['target'])\n    # y_pred = xgb.DMatrix(y_pred)\n    # y_true = xgb.DMatrix(y_true)\n\n    \n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:48:43.878768Z","iopub.execute_input":"2022-08-23T11:48:43.879836Z","iopub.status.idle":"2022-08-23T11:48:43.899213Z","shell.execute_reply.started":"2022-08-23T11:48:43.879788Z","shell.execute_reply":"2022-08-23T11:48:43.897605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric_np( target: np.ndarray,preds: np.ndarray) -> float:\n    indices = np.argsort(preds)[::-1]\n    preds, target = preds[indices], target[indices]\n\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n    d = np.sum(target[four_pct_mask]) / np.sum(target)\n\n    weighted_target = target * weight\n    lorentz = (weighted_target / weighted_target.sum()).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    n_pos = np.sum(target)\n    n_neg = target.shape[0] - n_pos\n    gini_max = 10 * n_neg * (n_pos + 20 * n_neg - 19) / (n_pos + 20 * n_neg)\n\n    g = gini / gini_max\n    # return 0.5 * (g + d)\n    return (0.5*g + 0.5*d)\n    \n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:48:43.902719Z","iopub.execute_input":"2022-08-23T11:48:43.903826Z","iopub.status.idle":"2022-08-23T11:48:43.918622Z","shell.execute_reply.started":"2022-08-23T11:48:43.903781Z","shell.execute_reply":"2022-08-23T11:48:43.916952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using amex metric to evaluate tabnet\nclass Amex_tabnet(Metric):\n    \n    def __init__(self):\n        self._name = 'amex_tabnet'\n        self._maximize = True\n\n    def __call__(self, y_true, y_pred):\n        amex = amex_metric_np(y_true, y_pred[:, 1])\n        return max(amex, 0.)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:48:43.921559Z","iopub.execute_input":"2022-08-23T11:48:43.922095Z","iopub.status.idle":"2022-08-23T11:48:43.937698Z","shell.execute_reply.started":"2022-08-23T11:48:43.922051Z","shell.execute_reply":"2022-08-23T11:48:43.935956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_amex_metric(y_true, y_pred):\n    return ('Score',\n            amex_metric(y_true, y_pred),\n            True)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:48:43.941310Z","iopub.execute_input":"2022-08-23T11:48:43.941845Z","iopub.status.idle":"2022-08-23T11:48:43.951172Z","shell.execute_reply.started":"2022-08-23T11:48:43.941799Z","shell.execute_reply":"2022-08-23T11:48:43.949929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Section","metadata":{}},{"cell_type":"markdown","source":"## Training Functions\\Model","metadata":{}},{"cell_type":"code","source":"def training_cv(model,X, y,name, split = 3,  train_all_data = False, users = df_sg[['customer_ID_']].to_pandas()):\n\n    kf = StratifiedKFold(n_splits=split, shuffle = True, random_state = CFG.seed)\n    amex_best = 0\n    acc_list = []\n    rec_list = []\n    auc_list = []\n    prec_list = []\n    amex_list = []\n    best_model = 0\n    gc.collect()\n    oof_complete = pd.DataFrame()\n    \n    if CFG.model_type == 'TABNET':\n        name = name+'_TABNET'\n    elif CFG.model_type == 'CATBOOST':\n        name = name+'_CATBOOST'\n    elif CFG.model_type == 'XGB':\n        name = name+'_XGB'\n    elif CFG.model_type == 'LOGIT':\n        name = name+'_LOGIT'\n    elif CFG.model_type == 'LGBM':\n        name = name+'_LGBM'\n        \n         \n    \n    for fold, (train_, val_df) in enumerate(( kf.split(X,y) )):\n        print('\\n')\n        print('#'*10)\n        print(f'Fold: {fold+1}/{split}')\n        print('#'*10)\n\n        if CFG.model_type == 'TABNET':\n            X_train, X_val = X.iloc[train_].values, X.iloc[val_df].values\n            y_train, y_val = y.iloc[train_].values, y.iloc[val_df].values\n\n            print(f'Train: {X_train.shape[0]} Test: {X_val.shape[0]}')\n            \n\n            model.fit(X_train,y_train,\n                    eval_set = [(X_train, y_train), (X_val,y_val)], \n                    eval_name = ['Train', 'Val'],\n                    max_epochs = CFG.tabEPOCS,\n                    batch_size = 2048,\n                    eval_metric = ['auc', 'accuracy', Amex_tabnet])\n            \n        elif CFG.model_type == 'XGB' or CFG.model_type == 'CATBOOST':\n\n\n\n            X_train, X_val = X.iloc[train_], X.iloc[val_df]\n            y_train, y_val = y.iloc[train_].values, y.iloc[val_df].values\n\n\n\n            print(f'Train: {X_train.shape[0]} Test: {X_val.shape[0]}')\n            \n            if CFG.model_type == 'CATBOOST':\n                eval = ((X_val, y_val))\n            elif CFG.model_type == 'XGB':\n                eval = [(X_train, y_train), (X_val,y_val)]\n        \n\n            model.fit(X_train,y_train,\n                    eval_set = eval, \n                    verbose = False)\n        else:\n\n            X_train, X_val = X.iloc[train_], X.iloc[val_df]\n            y_train, y_val = y.iloc[train_].values, y.iloc[val_df].values\n            print(f'Train: {X_train.shape[0]} Test: {X_val.shape[0]}')\n\n            model.fit(X_train,y_train,eval_metric=[lgb_amex_metric])\n        \n        pred = model.predict(X_val)\n        pred_prob = model.predict_proba(X_val)[:,1]\n        oof_df = users.iloc[val_df].copy()\n        oof_df['oof'] = pred_prob\n        oof_df['fold'] = fold\n        \n        oof_complete = pd.concat([oof_complete, oof_df])\n        \n        acc = accuracy_score(y_val,pred)\n        prec = precision_score(y_val,pred)\n        rec = recall_score(y_val,pred)\n        auc = roc_auc_score(y_val,pred)\n        amex_val = np.round(amex_metric(y_val,pred_prob),4)\n        amex_train = np.round(amex_metric(y_train, model.predict_proba(X_train)[:,1] ),4)\n\n        \n        print('Train:',amex_train,'Val:',amex_val)\n        save(model, f'{name}_fold{fold}_{CFG.seed}_{amex_train}_{amex_val}',CFG.models_path)\n        # oof_df.to_csv(f'{CFG.models_path}/{name}_fold{fold}_oof.csv')\n\n        \n        acc_list.append(acc)\n        prec_list.append(prec)\n        rec_list.append(rec)\n        auc_list.append(auc)\n        amex_list.append(amex_val)\n\n        print('\\n', classification_report(y_val, pred))\n\n        if amex_val > amex_best:\n            best_model = copy.deepcopy(model)\n            amex_best = amex_val\n\n        del X_train, X_val \n        del y_train, y_val, oof_df\n        _ = gc.collect()\n    \n    print('\\n')\n    print('%'*10)\n    print(f'OOFs AMEX - Folds')\n    print('%'*10)\n    oof_complete.to_csv(f'{CFG.models_path}/{name}/{name}_{CFG.seed}_oof_complete.csv')\n    amex_folds = np.round(amex_metric(label_df.sort_values(by='customer_ID')['target'].to_pandas().values, oof_complete.sort_values(by='customer_ID_')['oof'].values),4)\n    print(f'{bcolors.OKCYAN}Final Amex Metrix oof: {amex_folds}{bcolors.ENDC}')\n\n    #Plot - Graphics -------------------------------------------------------------------------------\n    if CFG.model_type == 'XGB' and False:\n        fig, ax = plt.subplots(ncols =2, figsize = (12,8))\n        print(list(best_model.evals_result()['validation_0'].keys()) )\n        \n        name_m = list(best_model.evals_result()['validation_0'].keys())[0]\n        ax[0].plot(best_model.evals_result()['validation_0'][f'{name_m}'], label='Train')\n        ax[0].plot(best_model.evals_result()['validation_1'][f'{name_m}'], label='Test')\n        ax[0].legend()\n        pred_prob = best_model.predict_proba(X)\n        ax[1].hist(pred_prob[:,1], bins = 30)\n        ax[0].set_ylabel(name_m)\n        plt.show()\n    \n    #Training BestModel in all Data\n    if train_all_data:\n        if CFG.model_type == 'TABNET':\n            best_model.fit(X.values,y.values,max_epochs = CFG.tabEPOCS_FULL)\n        else:\n            best_model.fit(X,y.values)\n        save(model, f'{name}',CFG.models_path,final = True)\n   \n    \n    \n    results = {'acc':acc_list,'precision':prec_list,'recall':rec_list,'auc':auc_list, 'amex':amex_list}\n    return best_model, results, amex_folds","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:48:43.953497Z","iopub.execute_input":"2022-08-23T11:48:43.954578Z","iopub.status.idle":"2022-08-23T11:48:44.284736Z","shell.execute_reply.started":"2022-08-23T11:48:43.954536Z","shell.execute_reply":"2022-08-23T11:48:44.283413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save Function","metadata":{}},{"cell_type":"code","source":"def save(model, name, path, final = False):\n    if final:\n        model_path = CFG.models_path\n    else:\n        model_path = path +'/'+name.split('_fold')[0]\n    if not os.path.exists(model_path):\n        os.makedirs(model_path)\n\n    # model.save_model(f'{model_path}/{name}.txt')\n    joblib.dump(model,os.path.join(model_path, f'{name}.sav'))\n    print(f'Saved {name} at {model_path}')","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:48:44.286784Z","iopub.execute_input":"2022-08-23T11:48:44.287936Z","iopub.status.idle":"2022-08-23T11:48:44.297606Z","shell.execute_reply.started":"2022-08-23T11:48:44.287870Z","shell.execute_reply":"2022-08-23T11:48:44.295462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tuning","metadata":{}},{"cell_type":"code","source":"#Tuning models\ndef tuning_models(models, parameters, X, y):\n    print('Initializing Tuning of models')\n    tuning_results = {}\n    for name, m in tqdm(models.items()):\n        print(f'>Tuning:{name}')\n\n        if CFG.type_tune == 'RANDOM':\n            clf = RandomizedSearchCV(m, parameters, \n                                     scoring = make_scorer(amex_metric_np,needs_proba=True), \n                                    #  scoring = 'f1',\n                                     random_state=CFG.seed, \n                                     cv = 5, \n                                     verbose = 10,\n                                     n_iter = 10\n                                     )\n        elif CFG.type_tune == 'HALVING':\n            clf = HalvingRandomSearchCV(m, parameters, \n                                    #  scoring = make_scorer(amex_metric_np,needs_proba=True), \n                                    #  scoring = 'f1',\n                                    random_state=CFG.seed, \n                                    cv = 5, \n                                    verbose = 10,\n                                    n_candidates = CFG.ite_tune\n                                    )\n\n\n        search = clf.fit(X, y)\n        tuning_results[name] = {'best_model':search.best_estimator_}\n        \n        \n    del X\n    return tuning_results[name]","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:48:44.300092Z","iopub.execute_input":"2022-08-23T11:48:44.300674Z","iopub.status.idle":"2022-08-23T11:48:44.314566Z","shell.execute_reply.started":"2022-08-23T11:48:44.300632Z","shell.execute_reply":"2022-08-23T11:48:44.312619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Models Initialization and Parameters","metadata":{}},{"cell_type":"code","source":"def init_models():\n    models = {}\n\n    for m in vars_groups.keys():\n        model_name = m\n\n###################################################################################################\n# TABNET\n###################################################################################################\n\n        if CFG.model_type == 'TABNET':\n            models[model_name] = TabNetClassifier(n_d = 32,\n                                    n_a = 64,\n                                    n_steps = 3,\n                                    gamma = 1.3,\n                                    #  cat_idxs = cat_index,                                 \n                                    n_independent = 2,\n                                    n_shared = 2,\n                                    momentum = 0.02,\n                                    clip_value = None,\n                                    lambda_sparse = 1e-3,\n                                    optimizer_fn = torch.optim.Adam,\n                                    scheduler_fn = torch.optim.lr_scheduler.CosineAnnealingLR,\n                                    scheduler_params = {\"T_max\" : 6},\n                                    mask_type = 'sparsemax',\n                                    seed = CFG.seed)\n            parameters = None\n\n###################################################################################################\n# XGB\n###################################################################################################\n\n\n        elif CFG.model_type == 'XGB':\n            if CFG.to_tune:\n\n                models[model_name] = xgb.XGBClassifier(objective ='binary:logistic')\n\n                parameters =  {'objective' :['binary:logistic'],\n                    'n_estimators':[300,500],\n                    'max_depth': [5,6, 10, 12], \n                    'colsample_bylevel':[0.1,0.3,0.5, 0.7],\n                    'learning_rate':list(np.linspace(0.001, 0.05, 20)),\n                    'subsample':[0.8, 0.9, 1],\n                    'reg_alpha':[10, 60, 100, 150, 200],\n                    'reg_lambda':[10,30],\n                    'tree_method':['gpu_hist'],\n                    'predictor': ['gpu_predictor'],\n                    'num_boost_round' : [15000],\n                    'random_state' : [CFG.seed],\n                    'gamma':[0,1,2],\n                    'min_child_weight':[1,5,10],\n                    'max_delta_step':[0,1,2],\n                    'boster':['dart']\n                                      }\n\n                \n\n            else:\n\n                parameters = {'base_score': 0.5,\n                                'booster': 'gbtree',\n                                'boster': 'dart',\n                                'colsample_bylevel': 0.7,\n                                'colsample_bynode': 1,\n                                'colsample_bytree': 1,\n                                'gamma': 1,\n                                'learning_rate': 0.02936842105263158,\n                                'max_delta_step': 0,\n                                'max_depth': 12,\n                                'min_child_weight': 5,\n                                'missing': None,\n                                'n_estimators': 500,\n                                'n_jobs': 1,\n                                'nthread': None,\n                                'num_boost_round': 9000,\n                                'predictor': 'gpu_predictor',\n                                'random_state': 19987,\n                                'reg_alpha': 10,\n                                'reg_lambda': 30,\n                                'scale_pos_weight': 1,\n                                'seed': None,\n                                'silent': None,\n                                'subsample': 0.9,\n                                'tree_method': 'gpu_hist',\n                                'verbosity': 1}\n\n                models[model_name] = xgb.XGBClassifier(objective ='binary:logistic',\n                                                       compute_importances=True,  \n                                                       **parameters)      \n\n            \n###################################################################################################\n# CATBOOST\n###################################################################################################          \n\n\n        elif CFG.model_type == 'CATBOOST':\n                \n\n                if CFG.to_tune:\n                    models[model_name] =  CatBoostClassifier(random_state=CFG.seed, verbose = False)\n\n                    # parameters = {'depth' : [4,5],\n                    # 'learning_rate' : [0.03],\n                    # 'iterations'    : [400],\n                    # 'l2_leaf_reg': [0,1,5],\n                    # 'bagging_temperature':[0,5],\n                    # 'task_type':[\"GPU\"]\n                    #     }\n\n                    parameters = {'depth' :list( np.arange(2,6)),\n                    'learning_rate' : list(np.linspace(0.001, 0.05, 10)),\n                    'iterations'    : [3000, 5000],\n                    'l2_leaf_reg': [0,2,10,12,15,30],\n                    # 'subsample': [0.6,0.8,0.9],\n                    'bagging_temperature':list(np.arange(0,10)),\n                    # 'num_leaves': [10, 20],\n                    'random_strength': list(np.linspace(0, 15, 5)),\n                    'task_type':[\"GPU\"]                        }\n\n                else:\n\n                    parameters = {'bagging_temperature': 1,\n                                    'depth': 6,\n                                    'iterations': 1000,\n                                    'l2_leaf_reg': 2,\n                                    'learning_rate': 0.020000000000000004,\n                                    'random_strength': 2.5}\n\n                    \n\n                    models[model_name] =  CatBoostClassifier(task_type=\"GPU\", \n                                                             random_state=CFG.seed, \n                                                             verbose = False, \n                                                             **parameters)\n                    \n###################################################################################################\n# LOGISTIC REGRESSION\n###################################################################################################\n\n        elif CFG.model_type == 'LOGIT':\n                \n                if CFG.to_tune:\n\n                    models[model_name] =  cuml.LogisticRegression(fit_intercept=True)\n\n\n                    parameters ={\n\n                        # 'solver' : ['newton-cg', 'lbfgs', 'liblinear'],\n                        'penalty' : ['l1', 'l2'],\n                        'C': [100, 10, 1.0, 0.1, 0.01],\n                        'max_iter':[1000,5000],\n                    }\n                else:\n                    models[model_name] =  cuml.LogisticRegression(fit_intercept=True,\n                                                                  C=0.1,\n                                                                  max_iter=5000, \n                                                                  penalty='l1')\n                    parameters = {}\n\n###################################################################################################\n# LGBM\n###################################################################################################\n\n        elif CFG.model_type == 'LGBM':\n            if CFG.to_tune:\n                parameters = {\n                    'lambda_l1': [0,0.08,0.1,0.2,1,10],\n                    'lambda_l2': [0, 0.08,0.1,0.2,1,10],\n                    'num_leaves': [50,100,250],\n                    'feature_fraction': [0.08,0.1,0.2,1],\n                    'bagging_fraction': [0.08,0.1,0.2,1],\n                    'bagging_freq': [2,4,6],\n                    'min_child_samples': [2,20,100], \n                    'max_depth': [10,30], \n                    'min_data_in_leaf': [1,5,10],\n                    'learning_rate': [0.005, 0.01, 0.02], \n                    'n_estimators':[300, 500],\n                    'drop_rate'  : [0.6], \n                    'max_drop' : [2,5,], \n                    'skip_drop' : [0.5, 0.8]}\n\n                models[model_name] = lgb.LGBMClassifier(boosting_type ='dart',\n                                                        xgboost_dart_mode = False, \n                                                        uniform_drop  = True, \n                                                        drop_seed = 666)\n            else:\n\n                # parameters = {\n                #     'max_depth': 5, \n                #     'min_data_in_leaf': 1,\n                #     'learning_rate': 0.01, \n                #     'lambda_l1': 1,\n                #     'lambda_l2': 1,\n                #     'num_leaves': 50,\n                #     'feature_fraction': 0.5,\n                #     'n_estimators':200}\n\n                parameters = {\n                        'objective': 'binary',\n                        'metric': \"binary_logloss\",\n                        'boosting': 'dart',\n                        'seed': CFG.seed,\n                        'num_leaves': 100,\n                        'feature_fraction': 0.20,\n                        'bagging_freq': 10,\n                        'bagging_fraction': 0.50,\n                        'n_jobs': -1,\n                        'lambda_l2': 2,\n                        'min_data_in_leaf': 40,\n                        # \"device_type\":\"gpu\",\n                                }\n\n\n                          \n                models[model_name] = lgb.LGBMClassifier(n_estimators =1200, \n                                                        learning_rate=0.03, reg_lambda=50,\n                                                        min_child_samples=2400,\n                                                        colsample_bytree=0.19,\n                                                        # device='gpu',\n                                                        random_state=42, **parameters)\n                \n\n    print('\\n')\n    print('#-'*10)\n    print(f'Variable Models to Run: {CFG.models_to_run}\\\n            \\nModel Selected: {CFG.model_type}\\\n           \\nSeed: {CFG.seed}\\\n           \\nTunning: {bcolors.t_o_f(CFG.to_tune)}\\\n           \\nParameters:{parameters}')\n    print('#-'*10)\n    print('\\n')\n    \n    return models, parameters","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:48:44.317816Z","iopub.execute_input":"2022-08-23T11:48:44.318760Z","iopub.status.idle":"2022-08-23T11:48:44.355168Z","shell.execute_reply.started":"2022-08-23T11:48:44.318713Z","shell.execute_reply":"2022-08-23T11:48:44.353488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training CFG","metadata":{}},{"cell_type":"code","source":"class CFG:\n    nome_pasta = 'RESULT_TESTE_OUTPUT'\n    \n    # models_path = '/content/models'\n    models_path =f'./{nome_pasta}'\n    t = 0.5\n    n_folds = 5\n\n    # MODELS AVAILABLE: CATBOOST | XGB | TABNET | LOGIT | LGBM\n    model_type = 'CATBOOST'\n    tabEPOCS = 60\n    tabEPOCS_FULL = 60\n\n    # MODELS VARIABLES - Select one or more in the Models Variables section ( Kaggle enviorment cant Train COMPLETO_MODEL- Out of Memory - Colab does)\n    models_to_run = ['LAST_MODEL', 'S_MODEL']\n\n    #Tunning\n    to_tune = False\n    type_tune = 'RANDOM'\n    ite_tune = 100\n\n    seeds = [202]\n    \n    #Train all data in the end\n    train_all_data = False\n\n    # Folder \n#     path_master = f'/content/drive/MyDrive/Kaggle/AMEX_2022/Models/'","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:48:44.357648Z","iopub.execute_input":"2022-08-23T11:48:44.358819Z","iopub.status.idle":"2022-08-23T11:48:44.375498Z","shell.execute_reply.started":"2022-08-23T11:48:44.358774Z","shell.execute_reply":"2022-08-23T11:48:44.373839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## RUN","metadata":{}},{"cell_type":"code","source":"\nbest_amex = 0\nbest_seed = 0\nbest_modelName = ''\nfor  ni, s in enumerate(CFG.seeds):\n    \n    print('\\n')\n    print('#%#'*30)\n    print(f'Evaluating: {bcolors.OKGREEN}({ni+1}/{len(CFG.seeds)}){bcolors.ENDC}- Best Amex Metrix so far:{best_amex} | with Seed: {best_seed}')\n    print('#%#'*30)\n\n    CFG.seed = s\n    models, parameters = init_models()\n\n    # print('Models Available:', list(models.keys()))\n\n\n    if CFG.models_to_run:\n        models_run = {nome:m for nome, m in models.items() if nome in CFG.models_to_run}\n    else:\n        models_run = models\n\n\n    results    = {}\n    best_model = {}\n    print('Training Models')\n    for mi, (name, model) in enumerate(models_run.items()):\n\n        print('\\n')\n        print('$%$'*30)\n        print(f'Evaluating Model {bcolors.OKGREEN}({mi+1}/{len(models_run)}): {(name)}{bcolors.ENDC} | Best Amex Metrix so far: {best_amex}  with Seed: {best_seed} of Model: {best_modelName}')\n        print('$%$'*30)\n\n\n        \n\n\n        if CFG.to_tune:\n            sample = np.random.randint(0,len(df_sg),400000)\n            tuning_results = tuning_models({name: models_run[name]}, parameters,df_sg[vars_groups[name]].iloc[sample].to_pandas().values , y.iloc[sample].values)\n            print('\\n')\n            print(tuning_results['best_model'].get_params())\n            print('\\n')\n        else:\n            tuning_results = {'best_model':models_run[name]}\n        \n        \n\n        best_model[(name)], metrics, amex_folds= training_cv(tuning_results['best_model'], df_sg[vars_groups[name]].to_pandas(), y, name, split = CFG.n_folds,train_all_data = CFG.train_all_data)\n        results[(name)] = {'best_model':best_model, 'accuracy': metrics['acc'],'recall': metrics['recall'],'precision': metrics['precision'], 'auc': metrics['auc'], 'amex':['amex']}\n\n        if amex_folds > best_amex:\n            best_amex = amex_folds\n            best_seed = s\n            best_modelName = name","metadata":{"execution":{"iopub.status.busy":"2022-08-23T11:48:44.377342Z","iopub.execute_input":"2022-08-23T11:48:44.379342Z","iopub.status.idle":"2022-08-23T11:54:54.567273Z","shell.execute_reply.started":"2022-08-23T11:48:44.379296Z","shell.execute_reply":"2022-08-23T11:54:54.565831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}