{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3739819,"sourceType":"datasetVersion","datasetId":2231132}],"dockerImageVersionId":30198,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook we build and train an XGBoost model using @raddar Kaggle dataset from [here][1] with discussion [here][2]. \n\n[1]: https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format\n[2]: https://www.kaggle.com/competitions/amex-default-prediction/discussion/328514","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np # CPU libraries\nimport matplotlib.pyplot as plt, gc, os","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:09.879521Z","iopub.execute_input":"2024-03-14T15:23:09.879954Z","iopub.status.idle":"2024-03-14T15:23:09.886037Z","shell.execute_reply.started":"2024-03-14T15:23:09.879919Z","shell.execute_reply":"2024-03-14T15:23:09.885089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" We will use [RAPIDS](https://rapids.ai/) and the GPU to create new features quickly.","metadata":{}},{"cell_type":"code","source":"import cupy, cudf # GPU libraries\n\nprint('RAPIDS version',cudf.__version__)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define constants\n\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\nLABEL_PATH = '../input/amex-default-prediction/train_labels.csv'\nTEST_PATH = '/kaggle/input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\n# Define VERSION for saved model files; incremented after each round of model training\nVER = 1\n\n# NaN value for missing integers\nNAN_VALUE = -127 # will fit in int8\n\n# as given in description\nCAT_VARS = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n\nTARGET_COLUMN = 'target'","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:09.887781Z","iopub.execute_input":"2024-03-14T15:23:09.888102Z","iopub.status.idle":"2024-03-14T15:23:09.897793Z","shell.execute_reply.started":"2024-03-14T15:23:09.888075Z","shell.execute_reply":"2024-03-14T15:23:09.897045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n\n    if usecols is not None: \n        df = cudf.read_parquet(path, columns=usecols)\n    else:\n        df = cudf.read_parquet(path)\n\n    # REDUCE DTYPE FOR CUSTOMER_ID by extracting last 16 chars of the hex string and then converting the data type to int64 which is only 4 bytes\n    df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    \n    # REDUCE DTYPE FOR DATE from string to datetime type\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    \n    # FILL NAN\n    df = df.fillna(NAN_VALUE) \n    # print('shape of data:', df.shape)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:09.898904Z","iopub.execute_input":"2024-03-14T15:23:09.899277Z","iopub.status.idle":"2024-03-14T15:23:09.912625Z","shell.execute_reply.started":"2024-03-14T15:23:09.899231Z","shell.execute_reply":"2024-03-14T15:23:09.911947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Reading train data...')\ntrain_df = read_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:09.913669Z","iopub.execute_input":"2024-03-14T15:23:09.914223Z","iopub.status.idle":"2024-03-14T15:23:11.732036Z","shell.execute_reply.started":"2024-03-14T15:23:09.914173Z","shell.execute_reply":"2024-03-14T15:23:11.731240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:11.734574Z","iopub.execute_input":"2024-03-14T15:23:11.734956Z","iopub.status.idle":"2024-03-14T15:23:11.984533Z","shell.execute_reply.started":"2024-03-14T15:23:11.734917Z","shell.execute_reply":"2024-03-14T15:23:11.983824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process and Feature Engineer Train Data\nWe will engineer features suggested by @huseyincot in his [notebook_1](https://www.kaggle.com/code/huseyincot/amex-catboost-0-793) and [notebook_2](https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created).","metadata":{}},{"cell_type":"code","source":"def feature_engineer(df, CAT_VARS=CAT_VARS):\n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    \n    cont_vars = [col for col in all_cols if col not in CAT_VARS]\n    \n    # creating addtional columns with aggregation of existing continuous var columns\n    cont_vars_agg = df.groupby(\"customer_ID\")[cont_vars].agg(['mean', 'std', 'min', 'max', 'last'])\n    # renaming the new agg cols with _\n    cont_vars_agg.columns = ['_'.join(x) for x in cont_vars_agg.columns]\n\n    cat_vars_agg = df.groupby(\"customer_ID\")[CAT_VARS].agg(['count', 'last', 'nunique'])\n    cat_vars_agg.columns = ['_'.join(x) for x in cat_vars_agg.columns]\n\n    df = cudf.concat([cont_vars_agg, cat_vars_agg], axis=1)\n    del cont_vars_agg, cat_vars_agg\n    print('shape after engineering', df.shape )\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:11.986032Z","iopub.execute_input":"2024-03-14T15:23:11.986535Z","iopub.status.idle":"2024-03-14T15:23:11.995858Z","shell.execute_reply.started":"2024-03-14T15:23:11.986492Z","shell.execute_reply":"2024-03-14T15:23:11.994982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = feature_engineer(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:11.996987Z","iopub.execute_input":"2024-03-14T15:23:11.997365Z","iopub.status.idle":"2024-03-14T15:23:13.052395Z","shell.execute_reply.started":"2024-03-14T15:23:11.997328Z","shell.execute_reply":"2024-03-14T15:23:13.051521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read labels\nlabel_df = cudf.read_csv(LABEL_PATH)\n\n# reduce data type\nlabel_df['customer_ID'] = label_df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\nlabel_df = label_df.set_index('customer_ID')\nprint(label_df.head(1))","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:13.053623Z","iopub.execute_input":"2024-03-14T15:23:13.054020Z","iopub.status.idle":"2024-03-14T15:23:13.838123Z","shell.execute_reply.started":"2024-03-14T15:23:13.053979Z","shell.execute_reply":"2024-03-14T15:23:13.837414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# merge the training features with correspoding labels in labels file\ntrain_df = train_df.merge(label_df, left_index=True, right_index=True, how='left')\ntrain_df.target = train_df.target.astype('int8')\ndel label_df\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain_df = train_df.sort_index().reset_index()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATURES\nFEATURE_COLUMNS = train_df.columns[1:-1]\nprint(f'There are {len(FEATURE_COLUMNS)} features: {FEATURE_COLUMNS}\\n')","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:13.840475Z","iopub.execute_input":"2024-03-14T15:23:13.840762Z","iopub.status.idle":"2024-03-14T15:23:13.845901Z","shell.execute_reply.started":"2024-03-14T15:23:13.840736Z","shell.execute_reply":"2024-03-14T15:23:13.845045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train\nWhen training with XGB, we use a special XGB dataloader called `DeviceQuantileDMatrix` which uses a small GPU memory footprint. This allows us to engineer more additional columns and train with more rows of data. ","metadata":{}},{"cell_type":"markdown","source":"Meaning of each parameter in the xgb_parms dictionary:\n\nmax_depth: It determines the maximum depth of each tree in the boosting process. Deeper trees can capture more complex patterns in the data, but may also lead to overfitting.\n\nlearning_rate: This parameter controls the step size at each iteration of boosting. A lower learning rate makes the model more robust by slowing down the learning process, while a higher learning rate can lead to faster convergence but may also cause the model to overshoot the optimal solution.\n\nsubsample: It specifies the fraction of samples used for training each tree. Setting it to less than 1.0 enables stochastic gradient boosting, where each tree is trained on a random subset of the training data. This can help prevent overfitting.\n\ncolsample_bytree: This parameter determines the fraction of features (columns) used for training each tree. It is similar to subsample, but instead of sampling rows, it samples columns. It helps to introduce randomness and reduce correlation between trees.\n\neval_metric: It specifies the evaluation metric used to assess the performance of the model during training. In this case, it is set to 'logloss', which is the logarithmic loss function commonly used for binary classification tasks.\n\nobjective: It defines the objective function to be optimized during training. Here, it is set to 'binary:logistic', indicating that the model is trained for binary classification using logistic regression as the base learner.\n\ntree_method: This parameter specifies the method used to build trees. 'gpu_hist' indicates that the tree construction algorithm is optimized for GPU computation, which can significantly speed up the training process when using GPU hardware.\n\npredictor: It specifies the method used for making predictions. 'gpu_predictor' indicates that the GPU should be used for making predictions, which can further improve inference speed.\n\nrandom_state: It sets the seed for random number generation, ensuring reproducibility of results. Setting it to a specific value (e.g., 42) ensures that the same results are obtained each time the model is trained with the same data and parameters.\n\nThese parameters collectively control various aspects of the XGBoost model training process, including the complexity of individual trees, the optimization strategy, and the hardware acceleration utilized. Adjusting these parameters carefully can help optimize model performance and generalization ability.\n\n","metadata":{}},{"cell_type":"code","source":"# LOAD XGB LIBRARY\nimport xgboost as xgb\nprint('XGB Version',xgb.__version__)\n\n# TRAIN RANDOM SEED\nSEED = 42\n\n# XGB MODEL PARAMETERS\nxgb_parms = { \n    'max_depth':4, \n    'learning_rate':0.05, \n    'subsample':0.8,\n    'colsample_bytree':0.6, \n    'eval_metric':'logloss',\n    'objective':'binary:logistic',\n    'tree_method':'gpu_hist',\n    'predictor':'gpu_predictor',\n    'random_state':SEED\n}","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:13.846964Z","iopub.execute_input":"2024-03-14T15:23:13.847305Z","iopub.status.idle":"2024-03-14T15:23:13.856456Z","shell.execute_reply.started":"2024-03-14T15:23:13.847265Z","shell.execute_reply":"2024-03-14T15:23:13.855563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost.core import DataIter\n\n# Define your custom iterator class by inheriting from xgb.core.DataIter\nclass DMatrixIter(DataIter):\n    def __init__(self, train_df=None, features=None, label=None, batch_size=256*1024):\n        super().__init__()\n        self.it = 0 # set iterator to 0\n\n        self.features = features\n        self.label = label\n        self.train_df = train_df\n        self.batch_size = batch_size\n        self.batches = int( np.ceil( len(train_df) / self.batch_size ) )\n\n    def reset(self):\n        '''Reset the iterator'''\n        self.it = 0\n\n    def next(self, input_data):\n        '''Yield next batch of data.'''\n        if self.it == self.batches:\n            return 0 # Return 0 when there's no more batch.\n        \n        # calculate the batch start row idx and end row idx based on batch size and the iteration value\n        start_idx = self.it * self.batch_size\n        end_idx = min( (self.it + 1) * self.batch_size, len(self.train_df) )\n        train_chunk = cudf.DataFrame(self.train_df.iloc[start_idx:end_idx])\n\n        # call input data on the training data chunk\n        input_data(data=train_chunk[self.features], label=train_chunk[self.label]) #, weight=dt['weight'])\n\n        # increment iterator for next batch\n        self.it += 1\n        return 1","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:13.857462Z","iopub.execute_input":"2024-03-14T15:23:13.857777Z","iopub.status.idle":"2024-03-14T15:23:13.867551Z","shell.execute_reply.started":"2024-03-14T15:23:13.857749Z","shell.execute_reply":"2024-03-14T15:23:13.866917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# metric defined by amex for accuracy calculation\ndef amex_metric_mod(y_true, y_pred):\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:23:13.868512Z","iopub.execute_input":"2024-03-14T15:23:13.868874Z","iopub.status.idle":"2024-03-14T15:23:13.879695Z","shell.execute_reply.started":"2024-03-14T15:23:13.868834Z","shell.execute_reply":"2024-03-14T15:23:13.878946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.to_pandas() # free GPU memory\n_=gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# K-fold cross-validation technique with XGB Classifier model using binary classifier as base learners\nfrom sklearn.model_selection import KFold\n\nFOLDS = 5\n\n# for calculating optional params to evaluate model predictions\nfeat_importances = [] # for storing features with corresponding importance values across each fold\noof = [] # out of fold prediction a.k.a prediction on validation set\n\n# initialise\nskf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\n\n# iterate over fold\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(train_df, train_df.target )):\n    \n    print('### Fold',fold+1)\n    print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n    \n    # TRAIN, VALID, TEST FOR FOLD K\n    Xy_train = DMatrixIter(train_df.loc[train_idx], FEATURE_COLUMNS, TARGET_COLUMN)\n\n    X_valid = train_df.loc[valid_idx, FEATURE_COLUMNS]\n    y_valid = train_df.loc[valid_idx, TARGET_COLUMN]\n    \n    # transform data into device quantile compatible data\n    dtrain = xgb.DeviceQuantileDMatrix(Xy_train, max_bin=256)\n    dvalid = xgb.DMatrix(data=X_valid, label=y_valid)\n    \n    # Train the model\n    model = xgb.train(xgb_parms, \n                dtrain=dtrain,\n                evals=[(dtrain,'train'),(dvalid,'valid')],\n                num_boost_round=9999,\n                early_stopping_rounds=100,\n                verbose_eval=100) \n    \n    # save the model with version\n    model.save_model(f'XGB_v{VER}_fold{fold}.xgb')\n    \n    # store feature importance based on weight assigned to each feature for current fold\n    feature_weights = model.get_score(importance_type='weight')\n    importance_df = pd.DataFrame({'feature':feature_weights.keys(),f'importance_{fold}':feature_weights.values()})\n    feat_importances.append(importance_df)\n            \n    # prediction on validation set\n    oof_preds = model.predict(dvalid)\n    # accuracy calculation\n    accuracy = amex_metric_mod(y_valid.values, oof_preds)\n    print('Kaggle Metric =',accuracy,'\\n')\n    \n    # save predictions\n    df = train_df.loc[valid_idx, ['customer_ID',TARGET_COLUMN] ].copy()\n    df['oof_pred'] = oof_preds\n    oof.append( df )\n    \n    # Note: very importanct to free memory after each fold\n    del dtrain, Xy_train, feature_weights, importance_df\n    del X_valid, y_valid, dvalid, model\n    _ = gc.collect()\n    \nprint('####################################')\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-03-14T15:23:13.882482Z","iopub.execute_input":"2024-03-14T15:23:13.882805Z","iopub.status.idle":"2024-03-14T15:32:23.924520Z","shell.execute_reply.started":"2024-03-14T15:23:13.882779Z","shell.execute_reply":"2024-03-14T15:32:23.923718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# overall calculation of accuracy\noof = pd.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\naccuracy = amex_metric_mod(oof.target.values, oof.oof_pred.values)\nprint('OVERALL Accuracy Metric =',accuracy)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CLEAN RAM\ndel train_df\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:32:23.925533Z","iopub.execute_input":"2024-03-14T15:32:23.925825Z","iopub.status.idle":"2024-03-14T15:32:24.129647Z","shell.execute_reply.started":"2024-03-14T15:32:23.925797Z","shell.execute_reply":"2024-03-14T15:32:24.128705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Visulalising OOF preds for training data - Optional","metadata":{}},{"cell_type":"code","source":"# to map oof to training data - we need to massage the training data and then map it back to oof\noof_xgb = pd.read_parquet(TRAIN_PATH, columns=['customer_ID']).drop_duplicates()\n\noof_xgb['customer_ID_hash'] = oof_xgb['customer_ID'].apply(lambda x: int(x[-16:],16) ).astype('int64')\noof_xgb = oof_xgb.set_index('customer_ID_hash')\n\noof_xgb = oof_xgb.merge(oof, left_index=True, right_index=True)\noof_xgb = oof_xgb.sort_index().reset_index(drop=True)\n\noof_xgb.to_csv(f'oof_xgb_v{VER}.csv',index=False)\noof_xgb.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:32:24.130946Z","iopub.execute_input":"2024-03-14T15:32:24.131414Z","iopub.status.idle":"2024-03-14T15:32:29.299610Z","shell.execute_reply.started":"2024-03-14T15:32:24.131375Z","shell.execute_reply":"2024-03-14T15:32:29.298777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot\nplt.hist(oof_xgb.oof_pred.values, bins=100)\nplt.title('OOF Predictions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:32:29.301227Z","iopub.execute_input":"2024-03-14T15:32:29.301724Z","iopub.status.idle":"2024-03-14T15:32:29.615007Z","shell.execute_reply.started":"2024-03-14T15:32:29.301680Z","shell.execute_reply":"2024-03-14T15:32:29.614222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clear RAM\ndel oof_xgb, oof\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:32:29.616316Z","iopub.execute_input":"2024-03-14T15:32:29.616735Z","iopub.status.idle":"2024-03-14T15:32:29.763327Z","shell.execute_reply.started":"2024-03-14T15:32:29.616695Z","shell.execute_reply":"2024-03-14T15:32:29.762383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Feature Importance - Optional","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nimp_feat_df = feat_importances[0].copy()\nfor k in range(1,FOLDS): imp_feat_df = imp_feat_df.merge(feat_importances[k], on='feature', how='left')\nimp_feat_df['importance'] = imp_feat_df.iloc[:,1:].mean(axis=1)\nimp_feat_df = imp_feat_df.sort_values('importance',ascending=False)\nimp_feat_df.to_csv(f'xgb_feature_importance_v{VER}.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:32:29.765083Z","iopub.execute_input":"2024-03-14T15:32:29.765624Z","iopub.status.idle":"2024-03-14T15:32:29.772791Z","shell.execute_reply.started":"2024-03-14T15:32:29.765584Z","shell.execute_reply":"2024-03-14T15:32:29.772067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_FEATURES = 20\nplt.figure(figsize=(10,5*NUM_FEATURES//10))\nplt.barh(np.arange(NUM_FEATURES,0,-1), imp_feat_df.importance.values[:NUM_FEATURES])\nplt.yticks(np.arange(NUM_FEATURES,0,-1), imp_feat_df.feature.values[:NUM_FEATURES])\nplt.title(f'XGB Feature Importance - Top {NUM_FEATURES}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:32:29.773859Z","iopub.execute_input":"2024-03-14T15:32:29.774555Z","iopub.status.idle":"2024-03-14T15:32:29.786047Z","shell.execute_reply.started":"2024-03-14T15:32:29.774516Z","shell.execute_reply":"2024-03-14T15:32:29.785330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process and Feature Engineer Test Data","metadata":{}},{"cell_type":"code","source":"print(f'Reading test data...')\ntest_df = read_file(path = TEST_PATH, usecols = ['customer_ID','S_2'])\ncustomer_ids = test_df[['customer_ID']].drop_duplicates().sort_index().values.flatten()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# COMPUTE SIZE OF 4 PARTS FOR TEST DATA\nNUM_PARTS = 4","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customer_ids = test_df[['customer_ID']].drop_duplicates().sort_index().values.flatten()\n\nchunk_size = len(customer_ids)//NUM_PARTS\nprint(f'We will process test data as {NUM_PARTS} separate parts.')\nprint(f'There will be {chunk_size} customers in each part (except the last part).')\nprint('Below are number of rows in each part:')\n\nrows = []\n\nfor k in range(NUM_PARTS):\n    if k==NUM_PARTS-1:\n        customer_ids_chunk = customer_ids[k*chunk_size:]\n    else:\n        customer_ids_chunk = customer_ids[k*chunk_size:(k+1)*chunk_size]\n    # shape return rows, columns: we only need rows   \n    rows_in_chunk = test_df.loc[test_df.customer_ID.isin(customer_ids_chunk)].shape[0]\n    # rows only have customer_ids information\n    rows.append(rows_in_chunk)\nprint(f\"number of chunks: {len(rows)} :: chunk size: {chunk_size}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"# INFER TEST DATA IN PARTS\nskip_rows = 0\nskip_cust = 0\ntest_preds = []\n\nfor k in range(NUM_PARTS):\n    \n    # 1.\n    print(f'\\nReading test data...')\n    test_df = read_file(path = TEST_PATH)\n    test_df = test_df.iloc[skip_rows:skip_rows+rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test_df.shape )\n    \n    # 2.\n    test_df = feature_engineer(test_df)\n    if k==NUM_PARTS-1:\n        test_df = test_df.loc[customer_ids[skip_cust:]]\n    else: \n        test_df = test_df.loc[customer_ids[skip_cust:skip_cust+chunk_size]]\n    skip_cust += chunk_size\n    \n    # 3. transform data for XGB\n    X_test = test_df[FEATURE_COLUMNS]\n    dtest = xgb.DMatrix(data=X_test)\n    test_df = test_df[['P_2_mean']] # reduce memory\n\n    del X_test\n    _=gc.collect()\n\n    # 4.\n    # initialise object\n    model = xgb.Booster()\n    # load object with fold 0 trained model for inference\n    model.load_model(f'XGB_v{VER}_fold0.xgb')\n    \n    # 5. predict \n    preds = model.predict(dtest)\n\n    # load model from other folds and predict\n    for f in range(1,FOLDS):\n        model.load_model(f'XGB_v{VER}_fold{f}.xgb')\n        preds = preds + model.predict(dtest)\n    preds = preds/FOLDS\n    \n    # 6. collect the overall predictions from different fold models\n    test_preds.append(preds)\n\n    del dtest, model\n    _ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:32:30.760223Z","iopub.execute_input":"2024-03-14T15:32:30.760534Z","iopub.status.idle":"2024-03-14T15:32:32.761637Z","shell.execute_reply.started":"2024-03-14T15:32:30.760505Z","shell.execute_reply":"2024-03-14T15:32:32.760517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit","metadata":{}},{"cell_type":"code","source":"test_preds = np.concatenate(test_preds)\ntest_df = cudf.DataFrame(index=customer_ids,data={'prediction':test_preds}","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:32:32.762555Z","iopub.status.idle":"2024-03-14T15:32:32.762909Z","shell.execute_reply.started":"2024-03-14T15:32:32.762744Z","shell.execute_reply":"2024-03-14T15:32:32.762761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = cudf.read_csv('../input/amex-default-prediction/sample_submission.csv')[['customer_ID']]\nsub['customer_ID_hash'] = sub['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\nsub = sub.set_index('customer_ID_hash')\nsub = sub.merge(test_df[['prediction']], left_index=True, right_index=True, how='left')\nsub = sub.reset_index(drop=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Visualise submission - Optional","metadata":{}},{"cell_type":"code","source":"sub.to_csv(f'submission.csv',index=False)\nprint('Submission file shape is', sub.shape )\nsub.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(sub.to_pandas().prediction, bins=100)\nplt.title('Test Predictions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T15:32:32.764324Z","iopub.status.idle":"2024-03-14T15:32:32.764642Z","shell.execute_reply.started":"2024-03-14T15:32:32.764488Z","shell.execute_reply":"2024-03-14T15:32:32.764504Z"},"trusted":true},"execution_count":null,"outputs":[]}]}