{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Load Libraries","metadata":{}},{"cell_type":"code","source":"ACTIVE = 'test'","metadata":{"execution":{"iopub.status.busy":"2022-08-24T07:09:08.464936Z","iopub.execute_input":"2022-08-24T07:09:08.465483Z","iopub.status.idle":"2022-08-24T07:09:08.495815Z","shell.execute_reply.started":"2022-08-24T07:09:08.465368Z","shell.execute_reply":"2022-08-24T07:09:08.494988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOAD LIBRARIES\nimport pandas as pd, numpy as np # CPU libraries\nimport matplotlib.pyplot as plt, gc, os\nimport joblib\n\nGPU = True\ntry:\n    import cupy, cudf\nexcept ImportError:\n    GPU = False\n\nif GPU:\n    print('RAPIDS version',cudf.__version__)\nelse:\n    print(\"Disabling cudf, using pandas instead\")\n    cudf = pd","metadata":{"execution":{"iopub.status.busy":"2022-08-24T07:09:08.497691Z","iopub.execute_input":"2022-08-24T07:09:08.498086Z","iopub.status.idle":"2022-08-24T07:09:12.216164Z","shell.execute_reply.started":"2022-08-24T07:09:08.498050Z","shell.execute_reply":"2022-08-24T07:09:12.214621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VERSION NAME FOR SAVED MODEL FILES\nVER = 1\nFEATURE_VER = 111\n\n# RANDOM SEED\nSEED = 108+5*VER+100*FEATURE_VER\n\n# FOLDS PER MODEL\nFOLDS = 10\n\n# NOTEBOOK PATH\nFEATURE_PATH = '../input/amex-preprocessed-test-dataset/'\nMODEL_PATH = '../input/amex-xgboost-pyramid-my-data/'\n\nprint(\"VER:\", VER)\nprint(\"fVER:\", FEATURE_VER)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T07:09:12.217528Z","iopub.execute_input":"2022-08-24T07:09:12.217919Z","iopub.status.idle":"2022-08-24T07:09:12.226187Z","shell.execute_reply.started":"2022-08-24T07:09:12.217884Z","shell.execute_reply":"2022-08-24T07:09:12.225292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data\nFeature engineering is all done in: https://www.kaggle.com/code/roberthatch/amex-feature-engg-gpu-or-cpu-process-in-chunks","metadata":{}},{"cell_type":"code","source":"if ACTIVE in ['train', 'all']:\n    print('Reading train data...')\n    TRAIN_PATH = f'{FEATURE_PATH}train_fe_v{FEATURE_VER}.parquet'\n    train = pd.read_parquet(TRAIN_PATH)\n    print(train.shape)\n\n    train = train.sample(frac=1, random_state=SEED)\n    train = train.reset_index(drop=True)\n    train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-24T07:09:12.228469Z","iopub.execute_input":"2022-08-24T07:09:12.229560Z","iopub.status.idle":"2022-08-24T07:09:12.235451Z","shell.execute_reply.started":"2022-08-24T07:09:12.229464Z","shell.execute_reply":"2022-08-24T07:09:12.234658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train XGB\nWe will train using `DeviceQuantileDMatrix`. This has a very small GPU memory footprint.","metadata":{}},{"cell_type":"code","source":"# LOAD XGB LIBRARY\nfrom sklearn.model_selection import KFold\nimport xgboost as xgb\nprint('XGB Version',xgb.__version__)\n\n\n# XGB MODEL PARAMETERS\nBASE_LEARNING_RATE = 0.01\nxgb_params = { \n    'max_depth': 7,\n    'subsample':0.75,\n    'colsample_bytree': 0.35,\n    'gamma':1.5,\n    'lambda':70,\n    'min_child_weight':8,\n\n    'objective':'binary:logistic',\n    'eval_metric':['logloss', 'auc'],  ## Early stopping is based on the last metric listed.\n    'tree_method':'gpu_hist',\n    'predictor':'gpu_predictor',\n    'random_state':SEED,\n\n    'num_parallel_tree':1\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-24T07:09:12.236862Z","iopub.execute_input":"2022-08-24T07:09:12.237532Z","iopub.status.idle":"2022-08-24T07:09:12.360985Z","shell.execute_reply.started":"2022-08-24T07:09:12.237499Z","shell.execute_reply":"2022-08-24T07:09:12.359525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NEEDED WITH DeviceQuantileDMatrix BELOW\nclass IterLoadForDMatrix(xgb.core.DataIter):\n    def __init__(self, df=None, features=None, target=None, batch_size=256*1024):\n        self.features = features\n        self.target = target\n        self.df = df\n        self.it = 0 # set iterator to 0\n        self.batch_size = batch_size\n        self.batches = int( np.ceil( len(df) / self.batch_size ) )\n        super().__init__()\n\n    def reset(self):\n        '''Reset the iterator'''\n        self.it = 0\n\n    def next(self, input_data):\n        '''Yield next batch of data.'''\n        if self.it == self.batches:\n            return 0 # Return 0 when there's no more batch.\n        \n        a = self.it * self.batch_size\n        b = min( (self.it + 1) * self.batch_size, len(self.df) )\n        dt = cudf.DataFrame(self.df.iloc[a:b])\n        input_data(data=dt[self.features], label=dt[self.target]) #, weight=dt['weight'])\n        self.it += 1\n        return 1","metadata":{"execution":{"iopub.status.busy":"2022-08-24T07:09:12.362214Z","iopub.execute_input":"2022-08-24T07:09:12.362591Z","iopub.status.idle":"2022-08-24T07:09:12.372094Z","shell.execute_reply.started":"2022-08-24T07:09:12.362555Z","shell.execute_reply":"2022-08-24T07:09:12.371332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## TODO, replace with newer version. Not done yet to preserve exact tested score.\n\n# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    print(\"  4%  :\", top_four)\n    print(\"  Gini:\", gini[1]/gini[0])\n    print(\"Kaggle:\", 0.5 * (gini[1]/gini[0] + top_four))\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T07:09:12.373608Z","iopub.execute_input":"2022-08-24T07:09:12.374331Z","iopub.status.idle":"2022-08-24T07:09:12.385431Z","shell.execute_reply.started":"2022-08-24T07:09:12.374297Z","shell.execute_reply":"2022-08-24T07:09:12.384426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"importances = []\nPYRAMID_W = [0.5, 2/3, 0.75, 0.875, 1, 0]\n\ndef run_training(train, features):\n    oof = []\n\n    skf = KFold(n_splits=FOLDS)\n    for fold,(train_idx, valid_idx) in enumerate(skf.split(\n                train, train.target )):\n        print('#'*25)\n        print('### Fold',fold+1)\n    \n        # TRAIN, VALID, TEST FOR FOLD K\n        X_valid = train.loc[valid_idx, features]\n        y_valid = train.loc[valid_idx, 'target']\n\n        print('### Train size',len(train_idx),'Valid size',len(valid_idx),'Valid positives',y_valid.sum())\n        print(f'### Training with all of fold data...')\n        print('#'*25)\n\n        dvalid = xgb.DMatrix(data=X_valid, label=y_valid)\n\n        # INFER XGB MODELS ON TEST DATA\n        print(\".\")\n        basic_score = 0\n        for (layer, w) in enumerate(PYRAMID_W[:-1]):\n            model = xgb.Booster()\n            model.load_model(f'{MODEL_PATH}XGB_v{VER}_fold{fold}_layer{layer}.xgb')\n            print(f'Loaded fold{fold}, layer{layer}')\n            ptest = model.predict(dvalid, output_margin=True)\n\n            ## reduce the impact of all model layers so far by w. This should be another way to reduce over-specialization, without the computational cost of DART\n            if (w < 1.0):\n                ptest = ptest * w\n\n            ## This set_base_margin is what informs the next layer of the prior training.\n            ## See code example from official demos: https://github.com/dmlc/xgboost/blob/master/demo/guide-python/boost_from_prediction.py\n            dvalid.set_base_margin(ptest)\n\n        layer = len(PYRAMID_W) - 1\n        model = xgb.Booster()\n        model.load_model(f'{MODEL_PATH}XGB_v{VER}_fold{fold}_layer{layer}.xgb')\n        # INFER OOF FOLD K\n        # Note: Not necessary with typical case ending pyramid with num_parallel_tree == 1, but more robust to divide best_ntree_limit by num_parallel_tree.\n        #   Oddly, iteration range is based only on num_boost_rounds, but best_ntree_limit is stored as num_boost_rounds * num_parallel_trees\n        print(\"Best_ntree_limit:\", model.best_ntree_limit//xgb_params['num_parallel_tree'])\n        oof_preds = model.predict(dvalid, iteration_range=(0,model.best_ntree_limit//xgb_params['num_parallel_tree']))\n        print('For this fold:')\n        ## TODO: update metric. Fork this notebook to confirm the latest version of the numpy implementation from author is even faster and equally accurate.\n        ## https://www.kaggle.com/code/rohanrao/amex-competition-metric-implementations\n        amex_metric_mod(y_valid.values, oof_preds)\n    \n        # SAVE OOF\n        df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n        df['oof_pred'] = oof_preds\n        oof.append( df )\n        \n        del X_valid, y_valid, dvalid, model\n        gc.collect()\n\n    print('#'*25)\n    print('OVERALL CV:')\n    oof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\n    amex_metric_mod(oof.target.values, oof.oof_pred.values)\n    return oof","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-08-24T07:09:12.387129Z","iopub.execute_input":"2022-08-24T07:09:12.388106Z","iopub.status.idle":"2022-08-24T07:09:12.402546Z","shell.execute_reply.started":"2022-08-24T07:09:12.388072Z","shell.execute_reply":"2022-08-24T07:09:12.401733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if ACTIVE in ['train', 'all']:\n    features = train.columns[1:-1]\n    print(f'There are {len(features)} features!')\n    print(train.shape)\n\n    oof = run_training(train, features)\n\n    # CLEAN RAM\n    del train\n    _ = gc.collect()\n\n    oof_xgb = pd.read_parquet(TRAIN_PATH, columns=['customer_ID']).drop_duplicates()\n    oof_xgb = oof_xgb.set_index('customer_ID')\n    oof_xgb = oof_xgb.merge(oof, left_index=True, right_index=True)\n    oof_xgb = oof_xgb.sort_index().reset_index(drop=True)\n    oof_xgb.to_csv(f'oof_xgb_v{VER}.csv',index=False)\n    oof_xgb.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-24T07:09:12.403973Z","iopub.execute_input":"2022-08-24T07:09:12.404573Z","iopub.status.idle":"2022-08-24T07:09:12.415034Z","shell.execute_reply.started":"2022-08-24T07:09:12.404540Z","shell.execute_reply":"2022-08-24T07:09:12.414285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Infer Test","metadata":{"_kg_hide-output":false,"_kg_hide-input":false}},{"cell_type":"code","source":"useless_features = [ \"R_1\", \"D_59\", \"S_11\", \"B_29\"]\n\nsummary = joblib.load('../input/amex-selected-features-v2/selected_features_v2.joblib')\nfeatures = summary['selected_features_names'] + summary['eliminated_features_names'][118:]\n\ndrop_cols = []\nfor col in features:\n    for col2 in useless_features:\n        if col2 in col:\n            drop_cols.append(col)\n\nfor col in drop_cols:\n    print(col)\n    features.remove(col)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T07:09:12.417863Z","iopub.execute_input":"2022-08-24T07:09:12.418277Z","iopub.status.idle":"2022-08-24T07:09:12.435688Z","shell.execute_reply.started":"2022-08-24T07:09:12.418241Z","shell.execute_reply":"2022-08-24T07:09:12.434986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if ACTIVE in ['test', 'all']:\n    gc.collect()\n\n    # INFER TEST DATA IN PARTS\n\n    TEST_SECTIONS = 2\n    TEST_SUB_SECTIONS = 2\n\n    test_preds = []\n    customers = False\n    for k in range(TEST_SECTIONS):\n        for i in range(TEST_SUB_SECTIONS):    \n            # READ PART OF TEST DATA\n            print(f'\\nReading test data...')\n            test = cudf.read_parquet(f'{FEATURE_PATH}test_{k}.parquet')\n            if i == 0:\n                print(f'=> Test part {k+1} has shape', test.shape )\n                if k == 0:\n                    customers = test['customer_ID'].to_arrow().to_pylist()\n                else:\n                    customers = customers + test['customer_ID'].to_arrow().to_pylist()\n\n            # TEST DATA FOR XGB\n            X_test = test[features]\n            n_rows = len(test.index)//TEST_SUB_SECTIONS\n            print(\".\")\n            if i+1 < TEST_SUB_SECTIONS:\n                X_test = X_test.iloc[i*n_rows:(i+1)*n_rows, :].copy()\n            elif TEST_SUB_SECTIONS > 1:\n                X_test = X_test.iloc[i*n_rows:, :].copy()\n            print(f'=> Test piece {k+1}, {i+1} has shape', X_test.shape )\n            del test\n            gc.collect()\n            dtest = xgb.DMatrix(data=X_test)\n            del X_test\n            gc.collect()\n            ## Need to reset to level 0 between folds.\n            reset_margin = dtest.get_base_margin()\n\n            # INFER XGB MODELS ON TEST DATA\n            print(\".\")\n            pred_folds = []\n            for f in range(FOLDS):\n                if (f > 0):\n                    dtest.set_base_margin(reset_margin)\n                for (layer, w) in enumerate(PYRAMID_W[:-1]):\n                    model = xgb.Booster()\n                    model.load_model(f'{MODEL_PATH}XGB_v{VER}_fold{f}_layer{layer}.xgb')\n                    print(f'Loaded fold{f}, layer{layer}')\n                    ptest = model.predict(dtest, output_margin=True)\n\n                    ## reduce the impact of all model layers so far by w. This should be another way to reduce over-specialization, without the computational cost of DART\n                    if (w < 1.0):\n                        ptest = ptest * w\n\n                    ## This set_base_margin is what informs the next layer of the prior training.\n                    ## See code example from official demos: https://github.com/dmlc/xgboost/blob/master/demo/guide-python/boost_from_prediction.py\n                    dtest.set_base_margin(ptest)\n\n                layer = len(PYRAMID_W) - 1\n                model = xgb.Booster()\n                model.load_model(f'{MODEL_PATH}XGB_v{VER}_fold{f}_layer{layer}.xgb')\n                print(\"Best_ntree_limit\", model.best_ntree_limit//xgb_params['num_parallel_tree'])\n                preds = model.predict(dtest, iteration_range=(0,model.best_ntree_limit//xgb_params['num_parallel_tree']))\n                \n                ## Create nested array to combine all predictions of a single fold together to rank them before averaging the predictions across folds.\n                if f == len(test_preds):\n                    test_preds.append([])\n                test_preds[f].append(preds)\n\n            # CLEAN MEMORY\n            del dtest, model, reset_margin\n            _ = gc.collect()","metadata":{"_kg_hide-output":false,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-08-24T07:13:46.640567Z","iopub.execute_input":"2022-08-24T07:13:46.641161Z","iopub.status.idle":"2022-08-24T07:14:03.377353Z","shell.execute_reply.started":"2022-08-24T07:13:46.641125Z","shell.execute_reply":"2022-08-24T07:14:03.375823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = [np.concatenate(p) for p in test_preds]\ntest_preds = np.mean(np.stack(preds), axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-24T07:19:41.222663Z","iopub.execute_input":"2022-08-24T07:19:41.223242Z","iopub.status.idle":"2022-08-24T07:19:41.228531Z","shell.execute_reply.started":"2022-08-24T07:19:41.223206Z","shell.execute_reply":"2022-08-24T07:19:41.227419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Submission CSV","metadata":{"_kg_hide-output":false,"_kg_hide-input":false}},{"cell_type":"code","source":"import pickle\n\nif ACTIVE in ['test', 'all']:\n    # WRITE SUBMISSION FILE\n    test = pd.DataFrame(index=customers,data={'prediction':test_preds})\n    sub = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')[['customer_ID']]\n    \n    with open('../input/aedp-starter-notebook/encoder_id.pkl', 'rb') as fin:\n        encoder = pickle.load(fin)\n        \n    sub['customer_ID_hash'] = encoder.transform(sub['customer_ID'])\n    sub = sub.set_index('customer_ID_hash')\n    \n    sub['prediction'] = test.loc[sub.index, 'prediction']\n    sub = sub.reset_index(drop=True)\n\n    # DISPLAY PREDICTIONS\n    sub.to_csv(f'submission.csv',index=False)\n    \n    print('Submission file shape is', sub.shape)\n    print(sub.isnull().sum())\n    sub.head()\n\n    # PLOT PREDICTIONS\n    plt.hist(sub.prediction, bins=100)\n    plt.title('Test Predictions')\n    plt.show()","metadata":{"_kg_hide-output":false,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2022-08-24T07:10:22.848153Z","iopub.status.idle":"2022-08-24T07:10:22.848774Z","shell.execute_reply.started":"2022-08-24T07:10:22.848549Z","shell.execute_reply":"2022-08-24T07:10:22.848570Z"},"trusted":true},"execution_count":null,"outputs":[]}]}