{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Originally copied from https://www.kaggle.com/code/cdeotte/xgboost-starter-0-793","metadata":{}},{"cell_type":"markdown","source":"# XGBoost Starter - LB 0.793\nIn this notebook we build and train an XGBoost model using @raddar Kaggle dataset from [here][1] with discussion [here][2]. Then we engineer features suggested by @huseyincot in his notebooks [here][3] and [here][4]. This XGB model achieves CV 0.792 LB 0.793! When training with XGB, we use a special XGB dataloader called `DeviceQuantileDMatrix` which uses a small GPU memory footprint. This allows us to engineer more additional columns and train with more rows of data. Our feature engineering is performed using [RAPIDS][5] on the GPU to create new features quickly.\n\n[1]: https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format\n[2]: https://www.kaggle.com/competitions/amex-default-prediction/discussion/328514\n[3]: https://www.kaggle.com/code/huseyincot/amex-catboost-0-793\n[4]: https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n[5]: https://rapids.ai/","metadata":{}},{"cell_type":"markdown","source":"# Load Libraries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom time import time\nfrom sklearn.model_selection import StratifiedKFold\nimport xgboost as xgb\nimport random\nimport joblib\n#from sklearn.pipeline import Pipeline\nimport gc #Coletor de lixo\nimport cudf\nimport cupy\nfrom cuml.metrics import roc_auc_score, log_loss\nfrom cuml.model_selection import train_test_split\n#from cuml.preprocessing import MinMaxScaler\n#from cuml.preprocessing import RobustScaler\n#from cuml.preprocessing import OneHotEncoder\n#from cuml import IncrementalPCA\n#from cuml.naive_bayes import BernoulliNB\n#import lightgbm as lgb\nimport matplotlib.pyplot as plt\nimport matplotlib.style as mplstyle #https://matplotlib.org/stable/gallery/style_sheets/style_sheets_reference.html\nfrom matplotlib.lines import Line2D  # for legend handl\nfrom IPython.display import FileLink\nplt.rcParams['agg.path.chunksize'] = 20000\n%matplotlib inline\n\ngc.enable()\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:33:01.009976Z","iopub.execute_input":"2022-08-11T16:33:01.010794Z","iopub.status.idle":"2022-08-11T16:33:02.525428Z","shell.execute_reply.started":"2022-08-11T16:33:01.010699Z","shell.execute_reply":"2022-08-11T16:33:02.524551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VERSION NAME FOR SAVED MODEL FILES\nVER = 30\n\n# TRAIN RANDOM SEED\nSEED = 21\n\n# FILL NAN VALUE\nNAN_VALUE = -127 # will fit in int8\n\n# FOLDS PER MODEL\nFOLDS = 5","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:33:02.526866Z","iopub.execute_input":"2022-08-11T16:33:02.527223Z","iopub.status.idle":"2022-08-11T16:33:02.532197Z","shell.execute_reply.started":"2022-08-11T16:33:02.527187Z","shell.execute_reply":"2022-08-11T16:33:02.530893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process and Feature Engineer Train Data\nWe will load @raddar Kaggle dataset from [here][1] with discussion [here][2]. Then we will engineer features suggested by @huseyincot in his notebooks [here][3] and [here][4]. We will use [RAPIDS][5] and the GPU to create new features quickly.\n\n[1]: https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format\n[2]: https://www.kaggle.com/competitions/amex-default-prediction/discussion/328514\n[3]: https://www.kaggle.com/code/huseyincot/amex-catboost-0-793\n[4]: https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n[5]: https://rapids.ai/","metadata":{}},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n    else: df = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n    #df = df.sort_values(['customer_ID','S_2'])\n    #df = df.reset_index(drop=True)\n    # FILL NAN\n    #df = df.fillna(NAN_VALUE) \n    print('shape of data:', df.shape)\n    \n    return df\n\nprint('Reading train data...')\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\ntrain = read_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:33:02.53404Z","iopub.execute_input":"2022-08-11T16:33:02.534407Z","iopub.status.idle":"2022-08-11T16:33:05.867552Z","shell.execute_reply.started":"2022-08-11T16:33:02.534371Z","shell.execute_reply":"2022-08-11T16:33:05.866708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"def process_and_feature_engineer(df):\n    # FEATURE ENGINEERING FROM \n    # https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_features]\n\n    test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n\n    test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n\n    df = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    del test_num_agg, test_cat_agg\n    print('shape after engineering', df.shape )\n    \n    return df\n\ntrain = process_and_feature_engineer(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T20:44:28.870839Z","iopub.execute_input":"2022-08-10T20:44:28.871212Z","iopub.status.idle":"2022-08-10T20:44:30.256462Z","shell.execute_reply.started":"2022-08-10T20:44:28.87117Z","shell.execute_reply":"2022-08-10T20:44:30.255671Z"}}},{"cell_type":"code","source":"from itertools import combinations as combi","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:33:05.869997Z","iopub.execute_input":"2022-08-11T16:33:05.870597Z","iopub.status.idle":"2022-08-11T16:33:05.875071Z","shell.execute_reply.started":"2022-08-11T16:33:05.870557Z","shell.execute_reply":"2022-08-11T16:33:05.874006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_and_feature_engineer(df, df_type = \"train\"):\n    t0 = time()\n    # FEATURE ENGINEERING MOSTLY FROM \n    # https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    # D_103 and D_109 are redundant: https://www.kaggle.com/code/raddar/redundant-features-amex\n    df = df.drop(['D_103','D_109'], axis = 1)\n    #B_29 has a strange behaviour on final months: https://www.kaggle.com/code/pavelvod/amex-eda-revealing-time-patterns-of-features/notebook\n    # And discussion https://www.kaggle.com/competitions/amex-default-prediction/discussion/328756\n    df = df.drop(['B_29'], axis = 1)\n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_features]\n    \n    #Dictionary with all dtypes\n    data_dictionary = {}\n    for col in all_cols:\n        data_dictionary[col] = df[col].dtype\n    \n    #low-cardinality\n    \n    if df_type == 'train':\n        global low_card\n        low_card = []\n        for col in num_features:\n            if data_dictionary[col] == 'int8':\n                #low-cardinality features\n                #Idea came from https://www.kaggle.com/competitions/amex-default-prediction/discussion/336557\n                low_card.append(col)\n    for col in low_card:\n        num_features.remove(col)\n        cat_features.append(col)\n            \n    if df_type == 'train':\n        #Greater than 90% missing values\n        #Code from https://www.kaggle.com/competitions/amex-default-prediction/discussion/336557\n        nullvals = df.isnull().sum() / df.shape[0]\n        global exclnullCols\n        exclnullCols = nullvals[nullvals>0.9].index.to_arrow().to_pylist()\n    \n    for col in exclnullCols:\n        #Keep only last statement\n        num_features.remove(col)\n        cat_features.append(col)\n        \n    # compute \"after pay\" features\n    # from https://www.kaggle.com/code/jiweiliu/rapids-cudf-feature-engineering-xgb\n    for bcol in [f'B_{i}' for i in [11,14,17]]+['D_39','D_131']+[f'S_{i}' for i in [16,23]]:\n        for pcol in ['P_2','P_3']:\n            if bcol in df.columns:\n                df[f'{bcol}-{pcol}'] = df[bcol] - df[pcol]\n                num_features.append(f'{bcol}-{pcol}')\n                all_cols.append(f'{bcol}-{pcol}')\n\n\n        \n    print(\"#Aggregating features - numerical\")\n    #print(num_features)\n    test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['last', 'mean','std','first'])\n    for col in test_num_agg.columns.levels[0]:\n        #test_num_agg[col,'amplitude1'] = test_num_agg[col,'max'] - test_num_agg[col,'min']\n        test_num_agg[col,'amplitude3'] = test_num_agg[col,'last'] - test_num_agg[col,'mean']\n        test_num_agg[col,'ampltiude2'] = test_num_agg[col,'last']/test_num_agg[col,'first']\n        #test_num_agg[col,'amplitude4'] = test_num_agg[col,'last'] - test_num_agg[col,'first']\n        #test_num_agg[col,'minmaxratio'] = (test_num_agg[col,'last']+1)/(test_num_agg[col,'mean']+1)\n        #test_num_agg[col,'firslasratio'] = (test_num_agg[col,'last']+1)/(test_num_agg[col,'first']+1)\n    \n        \n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n    for col in num_features:\n        test_num_agg.drop(col + '_' + 'first', axis = 1, inplace = True)\n        \n    test_num_agg = cudf.DataFrame(data = test_num_agg, columns = test_num_agg.columns, index = test_num_agg.index, dtype = 'float32')\n    \n    #XGBoost doesn't handle infinity values of amplitude2 column in a proper way\n    test_num_agg.replace([np.inf, -np.inf], np.nan, inplace=True)\n    \n    #NAN PROCESSING \n    if df_type == 'train':\n        global diction_cols_nan\n        diction_cols_nan = {}\n    \n    print(\"#NAN PROCESSING - numerical\")\n    for col in test_num_agg.columns:\n        if df_type == 'train':\n                diction_cols_nan[col] = (test_num_agg[col].min()-1).astype('float32')\n        test_num_agg[col] = test_num_agg[col].fillna(diction_cols_nan[col])\n    #test_num_agg = test_num_agg.fillna(-127)\n    #Dropping duplicate columns - numerical\n    print(\"Dropping Duplicates - numerical\")\n    if df_type == 'train':\n        global num_cols_to_keep\n        #Alternative to drop_duplicates\n        #Adapted from: https://stackoverflow.com/questions/14984119/python-pandas-remove-duplicate-columns\n        dups = [cc[1] for cc in combi(test_num_agg.columns, r=2) if (test_num_agg[cc[0]] == test_num_agg[cc[1]]).all()]\n        test_num_agg = test_num_agg.drop(dups, axis=1)\n        #test_num_agg = test_num_agg.T.drop_duplicates(keep = 'first').T\n        num_cols_to_keep = list(test_num_agg.columns)\n    test_num_agg = test_num_agg[num_cols_to_keep]\n    print(\"#Aggregating features - categorical\")\n    test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['last', 'count', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n    \n    #Some aggregations on test data will produce NaN\n    test_cat_agg = cudf.DataFrame(data = test_cat_agg, columns = test_cat_agg.columns, index = test_cat_agg.index)\n    \n    print(\"#NAN PROCESSING - categorical\")\n    #Fixing dtypes to avoid memory overconsumption\n    for col in test_cat_agg.columns:\n        for col2 in data_dictionary.keys():\n            if col.startswith(col2):\n                test_cat_agg[col] = test_cat_agg[col].astype(data_dictionary[col2])\n                if df_type == 'train':\n                    diction_cols_nan[col] = (test_cat_agg[col].min()-1).astype(data_dictionary[col2])\n                test_cat_agg[col] = test_cat_agg[col].fillna(diction_cols_nan[col])\n                break\n                \n        \n    print(\"Dropping Duplicates - categorical\")\n\n    if df_type == 'train':\n        #Dropping duplicate columns\n        global cat_cols_to_keep\n        #Alternative to drop_duplicates\n        #Adapted from: https://stackoverflow.com/questions/14984119/python-pandas-remove-duplicate-columns\n        dups = [cc[1] for cc in combi(test_cat_agg.columns, r=2) if (test_cat_agg[cc[0]] == test_cat_agg[cc[1]]).all()]\n        test_cat_agg = test_cat_agg.drop(dups, axis=1)\n        #test_cat_agg = test_cat_agg.T.drop_duplicates(keep = 'first').T\n        cat_cols_to_keep = list(test_cat_agg.columns)\n    test_cat_agg = test_cat_agg[cat_cols_to_keep]\n    #test_cat_agg = test_cat_agg.fillna(-127)\n\n    df = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    del test_num_agg, test_cat_agg\n    _ = gc.collect()\n    \n    #Dropping constant columns\n    if df_type == \"train\":\n        all_cols = df.columns\n        global constant_cols_to_remove\n        constant_cols_to_remove = []\n        for col in all_cols:\n            if df[col].nunique() == 1:\n                constant_cols_to_remove.append(col)\n        print(f\"Total number of constant columns to drop: {len(constant_cols_to_remove)}\")\n    df = df.drop(constant_cols_to_remove, axis=1)\n    \n    _ = gc.collect()\n    print('shape after engineering', df.shape )\n    t1 = time()\n    print(f\"Total engineering time: {(t1-t0):.3f}s\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:33:05.876698Z","iopub.execute_input":"2022-08-11T16:33:05.877161Z","iopub.status.idle":"2022-08-11T16:33:05.904739Z","shell.execute_reply.started":"2022-08-11T16:33:05.877092Z","shell.execute_reply":"2022-08-11T16:33:05.90382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = process_and_feature_engineer(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:33:05.907737Z","iopub.execute_input":"2022-08-11T16:33:05.90815Z","iopub.status.idle":"2022-08-11T16:34:29.172842Z","shell.execute_reply.started":"2022-08-11T16:33:05.908114Z","shell.execute_reply":"2022-08-11T16:34:29.171817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ADD TARGETS\ntargets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\ntargets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\ntargets = targets.set_index('customer_ID')\ntrain = train.merge(targets, left_index=True, right_index=True, how='left')\ntrain.target = train.target.astype('int8')\ndel targets\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain = train.sort_index().reset_index()\n\n# FEATURES\nFEATURES = train.columns[1:-1]\nprint(f'There are {len(FEATURES)} features!')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:34:29.175643Z","iopub.execute_input":"2022-08-11T16:34:29.175949Z","iopub.status.idle":"2022-08-11T16:34:30.330525Z","shell.execute_reply.started":"2022-08-11T16:34:29.175921Z","shell.execute_reply":"2022-08-11T16:34:30.329652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train XGB\nWe will train using `DeviceQuantileDMatrix`. This has a very small GPU memory footprint.","metadata":{}},{"cell_type":"code","source":"# LOAD XGB LIBRARY\nfrom sklearn.model_selection import KFold\nimport xgboost as xgb\nprint('XGB Version',xgb.__version__)\n\n# XGB MODEL PARAMETERS\nxgb_parms = { \n    'max_depth':4, \n    'learning_rate':0.05, \n    'subsample':0.8,\n    'colsample_bytree':0.6, \n    'eval_metric':'logloss',\n    'objective':'binary:logistic',\n    'tree_method':'gpu_hist',\n    'predictor':'gpu_predictor',\n    'random_state':SEED\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:34:30.331717Z","iopub.execute_input":"2022-08-11T16:34:30.332487Z","iopub.status.idle":"2022-08-11T16:34:30.339173Z","shell.execute_reply.started":"2022-08-11T16:34:30.332449Z","shell.execute_reply":"2022-08-11T16:34:30.338117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NEEDED WITH DeviceQuantileDMatrix BELOW\nclass IterLoadForDMatrix(xgb.core.DataIter):\n    def __init__(self, df=None, features=None, target=None, batch_size=256*1024):\n        self.features = features\n        self.target = target\n        self.df = df\n        self.it = 0 # set iterator to 0\n        self.batch_size = batch_size\n        self.batches = int( np.ceil( len(df) / self.batch_size ) )\n        super().__init__()\n\n    def reset(self):\n        '''Reset the iterator'''\n        self.it = 0\n\n    def next(self, input_data):\n        '''Yield next batch of data.'''\n        if self.it == self.batches:\n            return 0 # Return 0 when there's no more batch.\n        \n        a = self.it * self.batch_size\n        b = min( (self.it + 1) * self.batch_size, len(self.df) )\n        dt = cudf.DataFrame(self.df.iloc[a:b])\n        input_data(data=dt[self.features], label=dt[self.target]) #, weight=dt['weight'])\n        self.it += 1\n        return 1","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:34:30.340435Z","iopub.execute_input":"2022-08-11T16:34:30.340983Z","iopub.status.idle":"2022-08-11T16:34:30.351095Z","shell.execute_reply.started":"2022-08-11T16:34:30.340947Z","shell.execute_reply":"2022-08-11T16:34:30.350083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:34:30.352433Z","iopub.execute_input":"2022-08-11T16:34:30.352801Z","iopub.status.idle":"2022-08-11T16:34:30.363242Z","shell.execute_reply.started":"2022-08-11T16:34:30.352766Z","shell.execute_reply":"2022-08-11T16:34:30.362393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"importances = []\noof = []\ntrain = train.to_pandas() # free GPU memory\nTRAIN_SUBSAMPLE = 1.0\ngc.collect()\n\nskf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(\n            train, train.target )):\n    \n    # TRAIN WITH SUBSAMPLE OF TRAIN FOLD DATA\n    if TRAIN_SUBSAMPLE<1.0:\n        np.random.seed(SEED)\n        train_idx = np.random.choice(train_idx, \n                       int(len(train_idx)*TRAIN_SUBSAMPLE), replace=False)\n        np.random.seed(None)\n    \n    print('#'*25)\n    print('### Fold',fold+1)\n    print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n    print(f'### Training with {int(TRAIN_SUBSAMPLE*100)}% fold data...')\n    print('#'*25)\n    \n    # TRAIN, VALID, TEST FOR FOLD K\n    Xy_train = IterLoadForDMatrix(train.loc[train_idx], FEATURES, 'target')\n    X_valid = train.loc[valid_idx, FEATURES]\n    y_valid = train.loc[valid_idx, 'target']\n    \n    dtrain = xgb.DeviceQuantileDMatrix(Xy_train, max_bin=256)\n    dvalid = xgb.DMatrix(data=X_valid, label=y_valid)\n    \n    # TRAIN MODEL FOLD K\n    model = xgb.train(xgb_parms, \n                dtrain=dtrain,\n                evals=[(dtrain,'train'),(dvalid,'valid')],\n                num_boost_round=9999,\n                early_stopping_rounds=100,\n                verbose_eval=100) \n    model.save_model(f'XGB_v{VER}_fold{fold}.xgb')\n    \n    # GET FEATURE IMPORTANCE FOR FOLD K\n    dd = model.get_score(importance_type='weight')\n    df = pd.DataFrame({'feature':dd.keys(),f'importance_{fold}':dd.values()})\n    importances.append(df)\n            \n    # INFER OOF FOLD K\n    oof_preds = model.predict(dvalid)\n    acc = amex_metric_mod(y_valid.values, oof_preds)\n    print('Kaggle Metric =',acc,'\\n')\n    \n    # SAVE OOF\n    df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n    df['oof_pred'] = oof_preds\n    oof.append( df )\n    \n    del dtrain, Xy_train, dd, df\n    del X_valid, y_valid, dvalid, model\n    _ = gc.collect()\n    \nprint('#'*25)\noof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\nacc = amex_metric_mod(oof.target.values, oof.oof_pred.values)\nprint('OVERALL CV Kaggle Metric =',acc)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-11T16:34:30.364677Z","iopub.execute_input":"2022-08-11T16:34:30.36532Z","iopub.status.idle":"2022-08-11T16:42:43.995702Z","shell.execute_reply.started":"2022-08-11T16:34:30.365284Z","shell.execute_reply":"2022-08-11T16:42:43.99474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CLEAN RAM\ndel train\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:42:43.997125Z","iopub.execute_input":"2022-08-11T16:42:43.997936Z","iopub.status.idle":"2022-08-11T16:42:44.154221Z","shell.execute_reply.started":"2022-08-11T16:42:43.997897Z","shell.execute_reply":"2022-08-11T16:42:44.15314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save OOF Preds","metadata":{}},{"cell_type":"code","source":"oof_xgb = pd.read_parquet(TRAIN_PATH, columns=['customer_ID']).drop_duplicates()\noof_xgb['customer_ID_hash'] = oof_xgb['customer_ID'].apply(lambda x: int(x[-16:],16) ).astype('int64')\noof_xgb = oof_xgb.set_index('customer_ID_hash')\noof_xgb = oof_xgb.merge(oof, left_index=True, right_index=True)\noof_xgb = oof_xgb.sort_index().reset_index(drop=True)\noof_xgb.to_csv(f'oof_xgb_v{VER}.csv',index=False)\noof_xgb.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:42:44.157586Z","iopub.execute_input":"2022-08-11T16:42:44.158085Z","iopub.status.idle":"2022-08-11T16:42:48.250258Z","shell.execute_reply.started":"2022-08-11T16:42:44.158047Z","shell.execute_reply":"2022-08-11T16:42:48.249265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT OOF PREDICTIONS\nplt.hist(oof_xgb.oof_pred.values, bins=100)\nplt.title('OOF Predictions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:42:48.251697Z","iopub.execute_input":"2022-08-11T16:42:48.252091Z","iopub.status.idle":"2022-08-11T16:42:48.599879Z","shell.execute_reply.started":"2022-08-11T16:42:48.252053Z","shell.execute_reply":"2022-08-11T16:42:48.599067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CLEAR VRAM, RAM FOR INFERENCE BELOW\ndel oof_xgb, oof\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:42:48.601041Z","iopub.execute_input":"2022-08-11T16:42:48.601868Z","iopub.status.idle":"2022-08-11T16:42:48.762771Z","shell.execute_reply.started":"2022-08-11T16:42:48.601809Z","shell.execute_reply":"2022-08-11T16:42:48.76175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Importance","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndf = importances[0].copy()\nfor k in range(1,FOLDS): df = df.merge(importances[k], on='feature', how='left')\ndf['importance'] = df.iloc[:,1:].mean(axis=1)\ndf = df.sort_values('importance',ascending=False)\ndf.to_csv(f'xgb_feature_importance_v{VER}.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:42:48.764324Z","iopub.execute_input":"2022-08-11T16:42:48.766166Z","iopub.status.idle":"2022-08-11T16:42:48.79458Z","shell.execute_reply.started":"2022-08-11T16:42:48.766125Z","shell.execute_reply":"2022-08-11T16:42:48.793874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_FEATURES = 20\nplt.figure(figsize=(10,5*NUM_FEATURES//10))\nplt.barh(np.arange(NUM_FEATURES,0,-1), df.importance.values[:NUM_FEATURES])\nplt.yticks(np.arange(NUM_FEATURES,0,-1), df.feature.values[:NUM_FEATURES])\nplt.title(f'XGB Feature Importance - Top {NUM_FEATURES}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:42:48.79598Z","iopub.execute_input":"2022-08-11T16:42:48.796354Z","iopub.status.idle":"2022-08-11T16:42:49.064628Z","shell.execute_reply.started":"2022-08-11T16:42:48.796319Z","shell.execute_reply":"2022-08-11T16:42:49.063847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process and Feature Engineer Test Data\nWe will load @raddar Kaggle dataset from [here][1] with discussion [here][2]. Then we will engineer features suggested by @huseyincot in his notebooks [here][1] and [here][4]. We will use [RAPIDS][5] and the GPU to create new features quickly.\n\n[1]: https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format\n[2]: https://www.kaggle.com/competitions/amex-default-prediction/discussion/328514\n[3]: https://www.kaggle.com/code/huseyincot/amex-catboost-0-793\n[4]: https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n[5]: https://rapids.ai/","metadata":{}},{"cell_type":"code","source":"# CALCULATE SIZE OF EACH SEPARATE TEST PART\ndef get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k==NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    if verbose != '': print( rows )\n    return rows,chunk\n\n# COMPUTE SIZE OF 4 PARTS FOR TEST DATA\nNUM_PARTS = 4\nTEST_PATH = '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\nprint(f'Reading test data...')\ntest = read_file(path = TEST_PATH, usecols = ['customer_ID','S_2'])\ncustomers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\nrows,num_cust = get_rows(customers, test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:42:49.06586Z","iopub.execute_input":"2022-08-11T16:42:49.066687Z","iopub.status.idle":"2022-08-11T16:42:51.364916Z","shell.execute_reply.started":"2022-08-11T16:42:49.066642Z","shell.execute_reply":"2022-08-11T16:42:51.36409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer Test","metadata":{}},{"cell_type":"code","source":"# INFER TEST DATA IN PARTS\nskip_rows = 0\nskip_cust = 0\ntest_preds = []\n\nfor k in range(NUM_PARTS):\n    \n    # READ PART OF TEST DATA\n    print(f'\\nReading test data...')\n    test = read_file(path = TEST_PATH)\n    test = test.iloc[skip_rows:skip_rows+rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test.shape )\n    \n    # PROCESS AND FEATURE ENGINEER PART OF TEST DATA\n    test = process_and_feature_engineer(test, df_type = 'test')\n    if k==NUM_PARTS-1: test = test.loc[customers[skip_cust:]]\n    else: test = test.loc[customers[skip_cust:skip_cust+num_cust]]\n    skip_cust += num_cust\n    \n    # TEST DATA FOR XGB\n    X_test = test[FEATURES]\n    dtest = xgb.DMatrix(data=X_test)\n    test = test[['P_2_mean']] # reduce memory\n    del X_test\n    gc.collect()\n\n    # INFER XGB MODELS ON TEST DATA\n    model = xgb.Booster()\n    model.load_model(f'XGB_v{VER}_fold0.xgb')\n    preds = model.predict(dtest)\n    for f in range(1,FOLDS):\n        model.load_model(f'XGB_v{VER}_fold{f}.xgb')\n        preds += model.predict(dtest)\n    preds /= FOLDS\n    test_preds.append(preds)\n\n    # CLEAN MEMORY\n    del dtest, model\n    _ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:42:51.368579Z","iopub.execute_input":"2022-08-11T16:42:51.369311Z","iopub.status.idle":"2022-08-11T16:45:30.932521Z","shell.execute_reply.started":"2022-08-11T16:42:51.369274Z","shell.execute_reply":"2022-08-11T16:45:30.931687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Submission CSV","metadata":{}},{"cell_type":"code","source":"# WRITE SUBMISSION FILE\ntest_preds = np.concatenate(test_preds)\ntest = cudf.DataFrame(index=customers,data={'prediction':test_preds})\nsub = cudf.read_csv('../input/amex-default-prediction/sample_submission.csv')[['customer_ID']]\nsub['customer_ID_hash'] = sub['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\nsub = sub.set_index('customer_ID_hash')\nsub = sub.merge(test[['prediction']], left_index=True, right_index=True, how='left')\nsub = sub.reset_index(drop=True)\n\n# DISPLAY PREDICTIONS\nsub.to_csv(f'submission_xgb_v{VER}.csv',index=False)\nprint('Submission file shape is', sub.shape )\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:45:30.934033Z","iopub.execute_input":"2022-08-11T16:45:30.934465Z","iopub.status.idle":"2022-08-11T16:45:32.223756Z","shell.execute_reply.started":"2022-08-11T16:45:30.934411Z","shell.execute_reply":"2022-08-11T16:45:32.222635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT PREDICTIONS\nplt.hist(sub.to_pandas().prediction, bins=100)\nplt.title('Test Predictions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:45:32.227649Z","iopub.execute_input":"2022-08-11T16:45:32.22976Z","iopub.status.idle":"2022-08-11T16:45:32.960514Z","shell.execute_reply.started":"2022-08-11T16:45:32.229729Z","shell.execute_reply":"2022-08-11T16:45:32.959654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}