{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# LOAD LIBRARIES\nimport pandas as pd, numpy as np # CPU libraries\nimport cupy, cudf # GPU libraries\nimport matplotlib.pyplot as plt, gc, os\nfrom sklearn.model_selection import train_test_split\nprint('RAPIDS version',cudf.__version__)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-20T00:50:21.137348Z","iopub.execute_input":"2022-08-20T00:50:21.137808Z","iopub.status.idle":"2022-08-20T00:50:22.979854Z","shell.execute_reply.started":"2022-08-20T00:50:21.137710Z","shell.execute_reply":"2022-08-20T00:50:22.978903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VERSION NAME FOR SAVED MODEL FILES\nVER = 1\n\n# TRAIN RANDOM SEED\nSEED = 42\n\n# FILL NAN VALUE\nNAN_VALUE = -3 # will fit in int8\n\n# FOLDS PER MODEL\nFOLDS = 5","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:50:22.981756Z","iopub.execute_input":"2022-08-20T00:50:22.982262Z","iopub.status.idle":"2022-08-20T00:50:22.988098Z","shell.execute_reply.started":"2022-08-20T00:50:22.982223Z","shell.execute_reply":"2022-08-20T00:50:22.986740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n    else: df = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n    #df = df.sort_values(['customer_ID','S_2'])\n    #df = df.reset_index(drop=True)\n    # FILL NAN\n    df = df.fillna(NAN_VALUE) \n    print('shape of data:', df.shape)\n    \n    return df\n\nprint('Reading train data...')\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\ntrain = read_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:50:25.465690Z","iopub.execute_input":"2022-08-20T00:50:25.466112Z","iopub.status.idle":"2022-08-20T00:50:50.195729Z","shell.execute_reply.started":"2022-08-20T00:50:25.466078Z","shell.execute_reply":"2022-08-20T00:50:50.194635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:50:50.198301Z","iopub.execute_input":"2022-08-20T00:50:50.198960Z","iopub.status.idle":"2022-08-20T00:50:50.422199Z","shell.execute_reply.started":"2022-08-20T00:50:50.198905Z","shell.execute_reply":"2022-08-20T00:50:50.420984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[[\"Bn_38\",\"Bn_30\",\"Dn_114\"]] = train[[\"B_38\",\"B_30\",\"D_114\"]].astype(\"float\")","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:50:50.423878Z","iopub.execute_input":"2022-08-20T00:50:50.424296Z","iopub.status.idle":"2022-08-20T00:50:50.435562Z","shell.execute_reply.started":"2022-08-20T00:50:50.424254Z","shell.execute_reply":"2022-08-20T00:50:50.434327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[num_features_target]","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:50:50.450220Z","iopub.execute_input":"2022-08-20T00:50:50.450510Z","iopub.status.idle":"2022-08-20T00:50:50.529387Z","shell.execute_reply.started":"2022-08-20T00:50:50.450483Z","shell.execute_reply":"2022-08-20T00:50:50.528290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"positive_features_1 = ['B_9', 'D_75', 'D_58', 'B_7', 'B_23', 'B_3', 'B_4', 'B_16', 'B_1', 'Bn_38', 'B_37', 'B_19', 'B_20', 'B_11', 'R_1', 'D_74', 'B_22', 'Bn_30', 'B_8', 'R_3', 'D_55', 'R_2', 'D_41', 'P_4', 'R_10', 'B_28', 'R_4']","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:50:51.267451Z","iopub.execute_input":"2022-08-20T00:50:51.269244Z","iopub.status.idle":"2022-08-20T00:50:51.275353Z","shell.execute_reply.started":"2022-08-20T00:50:51.269202Z","shell.execute_reply":"2022-08-20T00:50:51.274149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"negative_features_1 = ['B_18', 'B_2', 'B_33', 'P_2', 'D_47', 'D_45', 'D_51', 'S_25', 'D_112']","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:50:51.334518Z","iopub.execute_input":"2022-08-20T00:50:51.335051Z","iopub.status.idle":"2022-08-20T00:50:51.345600Z","shell.execute_reply.started":"2022-08-20T00:50:51.334929Z","shell.execute_reply":"2022-08-20T00:50:51.344396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\ndef process_and_feature_engineer(df):\n    # FEATURE ENGINEERING FROM \n    # https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    #all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    df[[\"Bn_38\",\"Bn_30\",\"Dn_114\"]] = df[[\"B_38\",\"B_30\",\"D_114\"]].astype(\"float\")\n    df[\"positive_mean\"] = df[positive_features_1].mean(axis = 1)\n    df[\"positive_median\"] = df[positive_features_1].median(axis = 1)\n    df[\"positive_max\"] = df[positive_features_1].max(axis = 1)\n    df[\"positive_min\"] = df[positive_features_1].min(axis = 1)  \n    df[\"positive_std\"] = df[positive_features_1].std(axis = 1)  \n    df[\"positive_ratio_1\"] =(df[\"positive_mean\"] - df[\"positive_median\"])/df[\"positive_std\"] \n    df[\"positive_ratio_2\"] =(df[\"positive_mean\"] - df[\"positive_median\"])/df[\"positive_max\"] \n    df[\"negative_mean\"] = df[negative_features_1].mean(axis = 1)\n    df[\"negative_median\"] = df[negative_features_1].median(axis = 1)\n    df[\"negative_max\"] = df[negative_features_1].max(axis = 1)\n    df[\"negative_min\"] = df[negative_features_1].min(axis = 1)  \n    df[\"negative_std\"] = df[negative_features_1].std(axis = 1)  \n    df[\"negative_ratio_1\"] =(df[\"negative_mean\"] - df[\"negative_median\"])/df[\"negative_std\"] \n    df[\"negative_ratio_2\"] =(df[\"negative_mean\"] - df[\"negative_median\"])/df[\"negative_max\"] \n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    #num_features = ['B_18', 'B_2', 'B_9', 'D_75', 'B_33', 'D_58', 'P_2', 'B_7', 'B_23',\n        #'B_3', 'B_4', 'B_16', 'B_1', 'Bn_38', 'B_37', 'B_19', 'B_20', 'B_11', 'R_1', 'D_74', \n        #'B_22', 'Bn_30', 'B_8', 'R_3', 'D_55', 'D_47', \n        #'D_45', 'R_2', 'D_51', 'D_41', 'P_4', 'R_10', 'S_25',\n        #'B_28', 'R_4', 'D_112', 'D_44', 'S_15', 'Dn_114', 'R_27', 'D_127']\n    #num_features_encoding = []\n    #for num in range(len(num_features_target)):\n        #Total = list_Mean_encoded_subject_0[num]+list_Mean_encoded_subject_1[num]+\n        #       list_Mean_encoded_subject_2[num]+list_Mean_encoded_subject_3[num]+list_Mean_encoded_subject_4[num]\n        #df[num_features_target[num]+\"_mean_target\"] = df[num_features_target[num]].map( list_Mean_encoded_subject[num])\n        #num_features_encoding.append(num_features_target[num]+\"_mean_target\")       \n    cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_features+[\"target\", \"fold\"]]\n    #test_num_target[num_features+\"_mean_target\"] = df[num_features].map(means)\n    test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n    for col in test_num_agg.columns:\n            if 'last' in col and col.replace('last', 'first') in test_num_agg.columns:\n                test_num_agg[col + '_lag_sub'] = test_num_agg[col] - test_num_agg[col.replace('last', 'first')]\n                test_num_agg[col + '_lag_div'] = test_num_agg[col] / test_num_agg[col.replace('last', 'first')]\n        \n    diff_cols_a = [f\"B_{i}\" for i in [11, 14, 17]] + [\"D_39\", \"D_131\"] + [f\"S_{i}\" for i in [16, 23]]\n    diff_cols_b = [\"P_2\", \"P_3\"]\n    diff_feat = df[diff_cols_a + diff_cols_b + ['customer_ID']]\n    for a in diff_cols_a:\n            for b in diff_cols_b:\n                    diff_feat[f\"{a}-{b}\"] = diff_feat[a] - diff_feat[b]\n    diff_feat.drop(diff_cols_a + diff_cols_b, axis=1, inplace=True)\n    diff_feat = diff_feat.groupby('customer_ID').agg(['first', 'last','median','mean', 'std'])#.pipe(flatten_columns)\n    diff_feat.columns =  ['_'.join(x) for x in diff_feat.columns]\n    #dataset = cudf.concat([cat_feat, num_feat, diff_feat], axis=1)\n    test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n    \n    #test_num_deconding = df.groupby(\"customer_ID\")[num_features_encoding].agg(['mean', 'std', 'min', 'max', 'last'])\n    #test_num_deconding.columns = ['_'.join(x) for x in test_num_deconding.columns]\n    \n    #df = cudf.concat([test_num_agg, test_cat_agg,diff_feat ,test_num_deconding], axis=1)\n    df = cudf.concat([test_num_agg, test_cat_agg,diff_feat], axis=1)\n    del test_num_agg, test_cat_agg\n    print('shape after engineering', df.shape )\n    \n    return df\n\ntrain = process_and_feature_engineer(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:50:51.347635Z","iopub.execute_input":"2022-08-20T00:50:51.348040Z","iopub.status.idle":"2022-08-20T00:51:00.642161Z","shell.execute_reply.started":"2022-08-20T00:50:51.348003Z","shell.execute_reply":"2022-08-20T00:51:00.641109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ADD TARGETS\ntargets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\ntargets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\ntargets = targets.set_index('customer_ID')\ntrain = train.merge(targets, left_index=True, right_index=True, how='left')\ntrain.target = train.target.astype('int8')\ndel targets\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain = train.sort_index().reset_index()\n\n# FEATURES\nFEATURES = train.columns[1:-1]\nprint(f'There are {len(FEATURES)} features!')","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:51:00.643455Z","iopub.execute_input":"2022-08-20T00:51:00.644096Z","iopub.status.idle":"2022-08-20T00:51:02.306501Z","shell.execute_reply.started":"2022-08-20T00:51:00.644057Z","shell.execute_reply":"2022-08-20T00:51:02.305309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOAD XGB LIBRARY\nfrom sklearn.model_selection import KFold\nimport xgboost as xgb\nprint('XGB Version',xgb.__version__)\n\n# XGB MODEL PARAMETERS\n#{'n_estimators': 4743, 'max_depth': 9, 'learning_rate': 0.006558432723232228, \n#'gamma': 0.2, 'min_child_weight': 7, 'subsample': 0.6, \n#'colsample_bytree': 0.9, 'reg_alpha': 0.8, 'reg_lambda': 0.8}\n# XGB MODEL PARAMETERS\nxgb_parms = { \n    'max_depth':7, \n    'eta': 0.03,\n    'learning_rate':0.006558432723232228, \n    'subsample':0.88,\n    'colsample_bytree':0.5, \n    'eval_metric':'logloss',\n    'objective':'binary:logistic',\n    'tree_method':'gpu_hist',\n    'predictor':'gpu_predictor',\n    'random_state':SEED,\n    'gamma': 1.5,\n    'min_child_weight': 8,\n    'lambda': 70,\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:52:41.450583Z","iopub.execute_input":"2022-08-20T00:52:41.451681Z","iopub.status.idle":"2022-08-20T00:52:41.558994Z","shell.execute_reply.started":"2022-08-20T00:52:41.451631Z","shell.execute_reply":"2022-08-20T00:52:41.558010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NEEDED WITH DeviceQuantileDMatrix BELOW\nclass IterLoadForDMatrix(xgb.core.DataIter):\n    def __init__(self, df=None, features=None, target=None, batch_size=256*1024):\n        self.features = features\n        self.target = target\n        self.df = df\n        self.it = 0 # set iterator to 0\n        self.batch_size = batch_size\n        self.batches = int( np.ceil( len(df) / self.batch_size ) )\n        super().__init__()\n\n    def reset(self):\n        '''Reset the iterator'''\n        self.it = 0\n\n    def next(self, input_data):\n        '''Yield next batch of data.'''\n        if self.it == self.batches:\n            return 0 # Return 0 when there's no more batch.\n        \n        a = self.it * self.batch_size\n        b = min( (self.it + 1) * self.batch_size, len(self.df) )\n        dt = cudf.DataFrame(self.df.iloc[a:b])\n        input_data(data=dt[self.features], label=dt[self.target]) #, weight=dt['weight'])\n        self.it += 1\n        return 1","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:52:43.809116Z","iopub.execute_input":"2022-08-20T00:52:43.810653Z","iopub.status.idle":"2022-08-20T00:52:43.821291Z","shell.execute_reply.started":"2022-08-20T00:52:43.810595Z","shell.execute_reply":"2022-08-20T00:52:43.820169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:52:44.293330Z","iopub.execute_input":"2022-08-20T00:52:44.294195Z","iopub.status.idle":"2022-08-20T00:52:44.306642Z","shell.execute_reply.started":"2022-08-20T00:52:44.294155Z","shell.execute_reply":"2022-08-20T00:52:44.305150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"List_idx_rdm = cupy.load('../input/saved-list-random/list_idx_rdm_70.npy')","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:52:45.243017Z","iopub.execute_input":"2022-08-20T00:52:45.243414Z","iopub.status.idle":"2022-08-20T00:52:45.277220Z","shell.execute_reply.started":"2022-08-20T00:52:45.243381Z","shell.execute_reply":"2022-08-20T00:52:45.276140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.loc[train.index.isin(List_idx_rdm)]","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:52:46.546859Z","iopub.execute_input":"2022-08-20T00:52:46.547852Z","iopub.status.idle":"2022-08-20T00:52:47.873374Z","shell.execute_reply.started":"2022-08-20T00:52:46.547804Z","shell.execute_reply":"2022-08-20T00:52:47.872356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traind = train.loc[train.index.isin(List_idx_rdm)].copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:52:51.456363Z","iopub.execute_input":"2022-08-20T00:52:51.456734Z","iopub.status.idle":"2022-08-20T00:52:52.192261Z","shell.execute_reply.started":"2022-08-20T00:52:51.456703Z","shell.execute_reply":"2022-08-20T00:52:52.191183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traind","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:53:06.023322Z","iopub.execute_input":"2022-08-20T00:53:06.023893Z","iopub.status.idle":"2022-08-20T00:53:07.457069Z","shell.execute_reply.started":"2022-08-20T00:53:06.023849Z","shell.execute_reply":"2022-08-20T00:53:07.455995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traind.index = [i for i in range(len(traind.index))]","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:53:35.688953Z","iopub.execute_input":"2022-08-20T00:53:35.689622Z","iopub.status.idle":"2022-08-20T00:53:35.749263Z","shell.execute_reply.started":"2022-08-20T00:53:35.689585Z","shell.execute_reply":"2022-08-20T00:53:35.748183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"importances = []\noof = []\ntrain = traind.to_pandas() # free GPU memory\nTRAIN_SUBSAMPLE = 1.0\ngc.collect()\nfrom sklearn.model_selection import StratifiedKFold\n#skf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nskf = StratifiedKFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(\n            train, train.target )):\n    \n    # TRAIN WITH SUBSAMPLE OF TRAIN FOLD DATA\n    if TRAIN_SUBSAMPLE<1.0:\n        np.random.seed(SEED)\n        train_idx = np.random.choice(train_idx, \n                       int(len(train_idx)*TRAIN_SUBSAMPLE), replace=False)\n        np.random.seed(None)\n    \n    print('#'*25)\n    print('### Fold',fold+1)\n    print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n    print(f'### Training with {int(TRAIN_SUBSAMPLE*100)}% fold data...')\n    print('#'*25)\n    \n    # TRAIN, VALID, TEST FOR FOLD K\n    Xy_train = IterLoadForDMatrix(train.loc[train_idx], FEATURES, 'target')\n    X_valid = train.loc[valid_idx, FEATURES]\n    y_valid = train.loc[valid_idx, 'target']\n    \n    dtrain = xgb.DeviceQuantileDMatrix(Xy_train, max_bin=256)\n    dvalid = xgb.DMatrix(data=X_valid, label=y_valid)\n    \n    # TRAIN MODEL FOLD K\n    model = xgb.train(xgb_parms, \n                dtrain=dtrain,\n                evals=[(dtrain,'train'),(dvalid,'valid')],\n                num_boost_round=9999,\n                early_stopping_rounds=500,\n                verbose_eval=100) \n    model.save_model(f'XGB_v{VER}_fold{fold}.xgb')\n    \n    # GET FEATURE IMPORTANCE FOR FOLD K\n    dd = model.get_score(importance_type='weight')\n    df = pd.DataFrame({'feature':dd.keys(),f'importance_{fold}':dd.values()})\n    importances.append(df)\n            \n    # INFER OOF FOLD K\n    oof_preds = model.predict(dvalid)\n    acc = amex_metric_mod(y_valid.values, oof_preds)\n    print('Kaggle Metric =',acc,'\\n')\n    \n    # SAVE OOF\n    df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n    df['oof_pred'] = oof_preds\n    oof.append( df )\n    \n    del dtrain, Xy_train, dd, df\n    del X_valid, y_valid, dvalid, model\n    _ = gc.collect()\n    \nprint('#'*25)\noof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\nacc = amex_metric_mod(oof.target.values, oof.oof_pred.values)\nprint('OVERALL CV Kaggle Metric =',acc)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T00:53:38.226406Z","iopub.execute_input":"2022-08-20T00:53:38.226898Z","iopub.status.idle":"2022-08-20T01:33:31.334900Z","shell.execute_reply.started":"2022-08-20T00:53:38.226853Z","shell.execute_reply":"2022-08-20T01:33:31.333758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CLEAN RAM\ndel train\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:33:50.448577Z","iopub.execute_input":"2022-08-20T01:33:50.448957Z","iopub.status.idle":"2022-08-20T01:33:50.616818Z","shell.execute_reply.started":"2022-08-20T01:33:50.448909Z","shell.execute_reply":"2022-08-20T01:33:50.615658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_xgb = pd.read_parquet(TRAIN_PATH, columns=['customer_ID']).drop_duplicates()\noof_xgb['customer_ID_hash'] = oof_xgb['customer_ID'].apply(lambda x: int(x[-16:],16) ).astype('int64')\noof_xgb = oof_xgb.set_index('customer_ID_hash')\noof_xgb = oof_xgb.merge(oof, left_index=True, right_index=True)\noof_xgb = oof_xgb.sort_index().reset_index(drop=True)\noof_xgb.to_csv(f'oof_xgb_v{VER}.csv',index=False)\noof_xgb.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:33:51.916900Z","iopub.execute_input":"2022-08-20T01:33:51.917914Z","iopub.status.idle":"2022-08-20T01:33:55.798160Z","shell.execute_reply.started":"2022-08-20T01:33:51.917876Z","shell.execute_reply":"2022-08-20T01:33:55.797013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndf = importances[0].copy()\nfor k in range(1,FOLDS): df = df.merge(importances[k], on='feature', how='left')\ndf['importance'] = df.iloc[:,1:].mean(axis=1)\ndf = df.sort_values('importance',ascending=False)\ndf.to_csv(f'xgb_feature_importance_v{VER}.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:33:56.032046Z","iopub.execute_input":"2022-08-20T01:33:56.032778Z","iopub.status.idle":"2022-08-20T01:33:56.066252Z","shell.execute_reply.started":"2022-08-20T01:33:56.032735Z","shell.execute_reply":"2022-08-20T01:33:56.065204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_FEATURES = 20\nplt.figure(figsize=(10,5*NUM_FEATURES//10))\nplt.barh(np.arange(NUM_FEATURES,0,-1), df.importance.values[:NUM_FEATURES])\nplt.yticks(np.arange(NUM_FEATURES,0,-1), df.feature.values[:NUM_FEATURES])\nplt.title(f'XGB Feature Importance - Top {NUM_FEATURES}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:34:00.134726Z","iopub.execute_input":"2022-08-20T01:34:00.135143Z","iopub.status.idle":"2022-08-20T01:34:00.479409Z","shell.execute_reply.started":"2022-08-20T01:34:00.135108Z","shell.execute_reply":"2022-08-20T01:34:00.478395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT OOF PREDICTIONS\nplt.hist(oof_xgb.oof_pred.values, bins=100)\nplt.title('OOF Predictions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:34:07.879104Z","iopub.execute_input":"2022-08-20T01:34:07.880156Z","iopub.status.idle":"2022-08-20T01:34:08.245037Z","shell.execute_reply.started":"2022-08-20T01:34:07.880103Z","shell.execute_reply":"2022-08-20T01:34:08.243913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CLEAR VRAM, RAM FOR INFERENCE BELOW\ndel oof_xgb, oof\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:34:30.983292Z","iopub.execute_input":"2022-08-20T01:34:30.983658Z","iopub.status.idle":"2022-08-20T01:34:31.169421Z","shell.execute_reply.started":"2022-08-20T01:34:30.983628Z","shell.execute_reply":"2022-08-20T01:34:31.168187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CALCULATE SIZE OF EACH SEPARATE TEST PART\ndef get_rows(customers, test, NUM_PARTS = 20, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k==NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    if verbose != '': print( rows )\n    return rows,chunk\n\n# COMPUTE SIZE OF 4 PARTS FOR TEST DATA\nNUM_PARTS = 20\nTEST_PATH = '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\nprint(f'Reading test data...')\ntest = read_file(path = TEST_PATH, usecols = ['customer_ID','S_2'])\ncustomers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\nrows,num_cust = get_rows(customers, test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:34:32.686864Z","iopub.execute_input":"2022-08-20T01:34:32.687588Z","iopub.status.idle":"2022-08-20T01:34:36.735128Z","shell.execute_reply.started":"2022-08-20T01:34:32.687549Z","shell.execute_reply":"2022-08-20T01:34:36.734017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# original code\ndef read_file_GPU(path = '', usecols = None):\n    # read_parquet() function can read the parquet-type file.\n    # if you want to specify columns:\n    if usecols is not None: \n        df = cudf.read_parquet(path, columns=usecols)\n    # if you want to read all columns:\n    else: df = cudf.read_parquet(path)\n    \n    df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    df = df.fillna(NAN_VALUE) \n    print('shape of data:', df.shape)\n    \n    return df\n\n# modified code(CPU, batch load)\ndef read_file_CPU(path = '', iter_batch = None, usecols = None):\n    if usecols is not None:\n        # when retrieving only some columns(1~3), there is no problem even if all rows are retrieved.\n        df = pd.read_parquet(path, columns=usecols)\n    else:\n        # When importing all columns, data is imported in batch format.\n        df = iter_batch\n    \n    # it performs the same processing as it did with cudf.\n    df['customer_ID'] = df['customer_ID'].apply(lambda x : int(x[-16:],16)).astype('int64') \n    df.S_2 = pd.to_datetime( df.S_2 )\n    df.fillna(NAN_VALUE, inplace=True)\n    print('shape of data:', df.shape)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:34:42.648700Z","iopub.execute_input":"2022-08-20T01:34:42.649239Z","iopub.status.idle":"2022-08-20T01:34:42.664503Z","shell.execute_reply.started":"2022-08-20T01:34:42.649186Z","shell.execute_reply":"2022-08-20T01:34:42.662390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_PATH = '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\ntest = read_file_CPU(path = TEST_PATH, usecols = ['customer_ID','S_2'])\ntest","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:34:42.879379Z","iopub.execute_input":"2022-08-20T01:34:42.880165Z","iopub.status.idle":"2022-08-20T01:35:47.620774Z","shell.execute_reply.started":"2022-08-20T01:34:42.880121Z","shell.execute_reply":"2022-08-20T01:35:47.619585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\ncustomers","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:35:47.623019Z","iopub.execute_input":"2022-08-20T01:35:47.623482Z","iopub.status.idle":"2022-08-20T01:35:48.004019Z","shell.execute_reply.started":"2022-08-20T01:35:47.623443Z","shell.execute_reply":"2022-08-20T01:35:48.002876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_rows(customers, test, NUM_PARTS = 10, verbose = ''):\n    # One chunk(size of PART) can be obtained by dividing the size of the entire test-dataset by the number of PARTs.\n    # It is similar to finding the batch size when training the model.\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        # Keep the remain to cc if this PART is the last one.\n        if k==NUM_PARTS-1: \n            cc = customers[k*chunk:]\n        # If not the last PART, cut it in chunks from the front and put it in cc.\n        else: \n            cc = customers[k*chunk:(k+1)*chunk]\n        # Calculate the PART size by finding the number of customer_IDs included in the current PART.\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        # rows contain the size of 10 PARTs.\n        rows.append(s)\n    if verbose != '': print( rows )\n    return rows, chunk","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:35:48.005575Z","iopub.execute_input":"2022-08-20T01:35:48.006470Z","iopub.status.idle":"2022-08-20T01:35:48.016088Z","shell.execute_reply.started":"2022-08-20T01:35:48.006438Z","shell.execute_reply":"2022-08-20T01:35:48.014823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# In the original code, the test dataset was divided into four.\n# But in case of using a kaggle server, you will get a GPU memory overflow error when you specify 4r, so we will divide it into 10.\nNUM_PARTS = 10\nrows,num_cust = get_rows(customers, test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:35:48.019220Z","iopub.execute_input":"2022-08-20T01:35:48.019718Z","iopub.status.idle":"2022-08-20T01:35:49.361648Z","shell.execute_reply.started":"2022-08-20T01:35:48.019674Z","shell.execute_reply":"2022-08-20T01:35:49.360382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:35:49.363309Z","iopub.execute_input":"2022-08-20T01:35:49.363974Z","iopub.status.idle":"2022-08-20T01:35:49.540975Z","shell.execute_reply.started":"2022-08-20T01:35:49.363910Z","shell.execute_reply":"2022-08-20T01:35:49.539682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyarrow.parquet import ParquetFile","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:35:49.542524Z","iopub.execute_input":"2022-08-20T01:35:49.543550Z","iopub.status.idle":"2022-08-20T01:35:49.552212Z","shell.execute_reply.started":"2022-08-20T01:35:49.543505Z","shell.execute_reply":"2022-08-20T01:35:49.551078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skip_rows = 0\nskip_cust = 0\ntest_preds = []\n\n# Create an iteration object. By calling the object, you can raise as many rows as PART units to the CPU memory.\nTEST_PATH = '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\nbatch = ParquetFile(TEST_PATH)\n# Since batch_size is fixed at one time call, it is not possible to get a different PARt size each time.\n# Because of this, the customer_ID that should be predicted in the current batch is possible to already be loaded in the previous batch.\n# So, set the batch size to the largest PART size, and merge it with the previous PART so that all customer_IDs can be indexed.\nprev_batch = pd.DataFrame()\nfor pres_batch, k in zip(batch.iter_batches(batch_size=max(rows)), range(NUM_PARTS)): \n    print(f'\\nReading test data...')\n    # Merge the dataset loaded from the previous batch and the current batch.\n    iter_batch = pd.concat([prev_batch, pres_batch.to_pandas()])\n    test_cpu = read_file_CPU(iter_batch = iter_batch, path = TEST_PATH)\n    # Then, Overwitten the previous batch with the current(present) batch.\n    # And clear the memory.\n    prev_batch = pres_batch.to_pandas()\n    del pres_batch, iter_batch\n    _ = gc.collect()\n    \n    # Load the PART into the GPU memory for pre-processing.\n    test_gpu = cudf.DataFrame(test_cpu)\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test_gpu.shape)\n    \n    # With preprocessing, statistics are obtained for each customer_ID, and duplicate customer_IDs are removed.\n    # process_and_feature_engineer() uses cudf internally, not pandas.\n    # So, when preprocessing, the data must be placed on the GPU.\n    test_gpu = process_and_feature_engineer(test_gpu)\n    \n    # num_cust is the chunk size for de-duplicated customer_ID.\n    # We will do the credit default prediction for all customer_IDs by increasing the chunk size.\n    if k==NUM_PARTS-1: \n        test_gpu = test_gpu.loc[customers[skip_cust:]]\n    else: \n        test_gpu = test_gpu.loc[customers[skip_cust:skip_cust+num_cust]]\n    skip_cust += num_cust\n    print('shape after indexing(by chunk size)', test_gpu.shape)\n    \n    # Pass the features excluding the label from the test data to X_test.\n    X_test = test_gpu[FEATURES]\n    dtest = xgb.DMatrix(data=X_test)\n    del X_test\n    gc.collect()\n\n    # With our trained model, we can predict default.\n    # Our model was trained by cross-validation.\n    # Prediction is perfromed in the same way, and the prediction result can be obtained as the average value of all FOLD results.\n    model = xgb.Booster()\n    model.load_model(f'XGB_v{VER}_fold0.xgb')\n    preds = model.predict(dtest)\n    for f in range(1,FOLDS):\n        model.load_model(f'XGB_v{VER}_fold{f}.xgb')\n        preds += model.predict(dtest)\n    preds /= FOLDS\n    test_preds.append(preds)\n\n    # free the memory\n    del dtest, model\n    _ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:35:49.553781Z","iopub.execute_input":"2022-08-20T01:35:49.555282Z","iopub.status.idle":"2022-08-20T01:50:57.570012Z","shell.execute_reply.started":"2022-08-20T01:35:49.555188Z","shell.execute_reply":"2022-08-20T01:50:57.569005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# WRITE SUBMISSION FILE\ntest_preds = np.concatenate(test_preds)\ntest = cudf.DataFrame(index=customers,data={'prediction':test_preds})\nsub = cudf.read_csv('../input/amex-default-prediction/sample_submission.csv')[['customer_ID']]\nsub['customer_ID_hash'] = sub['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\nsub = sub.set_index('customer_ID_hash')\nsub = sub.merge(test[['prediction']], left_index=True, right_index=True, how='left')\nsub = sub.reset_index(drop=True)\n\n# DISPLAY PREDICTIONS\nsub.to_csv(f'submission_xgb_v{VER}.csv',index=False)\nprint('Submission file shape is', sub.shape )\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:50:57.572652Z","iopub.execute_input":"2022-08-20T01:50:57.573058Z","iopub.status.idle":"2022-08-20T01:50:58.823030Z","shell.execute_reply.started":"2022-08-20T01:50:57.573016Z","shell.execute_reply":"2022-08-20T01:50:58.821915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PLOT OOF PREDICTIONS\nplt.hist(sub.prediction.to_arrow(), bins=100)\nplt.title('submission Predictions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:51:11.354168Z","iopub.execute_input":"2022-08-20T01:51:11.354762Z","iopub.status.idle":"2022-08-20T01:51:11.732523Z","shell.execute_reply.started":"2022-08-20T01:51:11.354726Z","shell.execute_reply":"2022-08-20T01:51:11.731520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.prediction.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:51:52.139362Z","iopub.execute_input":"2022-08-20T01:51:52.139721Z","iopub.status.idle":"2022-08-20T01:51:52.174205Z","shell.execute_reply.started":"2022-08-20T01:51:52.139690Z","shell.execute_reply":"2022-08-20T01:51:52.173124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}