{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3739819,"sourceType":"datasetVersion","datasetId":2231132}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nXGB model with GPU usage by cudf using parameter obtained from Grid Search\n\n\nScored 0.80322 for competition's metrics","metadata":{}},{"cell_type":"markdown","source":"# Load Libraries","metadata":{}},{"cell_type":"code","source":"# LOAD LIBRARIES\nimport pandas as pd, numpy as np # CPU libraries\nimport cupy, cudf # GPU libraries\nimport matplotlib.pyplot as plt, gc, os\nimport seaborn as sns\nimport gc\nimport torch\nfrom numba import cuda\n\nprint('RAPIDS version',cudf.__version__)\npd.set_option('display.max_columns', None)","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:11.52155Z","iopub.execute_input":"2025-04-24T18:39:11.52181Z","iopub.status.idle":"2025-04-24T18:39:22.929606Z","shell.execute_reply.started":"2025-04-24T18:39:11.521783Z","shell.execute_reply":"2025-04-24T18:39:22.928842Z"},"trusted":true},"outputs":[{"name":"stdout","text":"RAPIDS version 25.02.02\n","output_type":"stream"}],"execution_count":1},{"cell_type":"code","source":"# VERSION NAME FOR SAVED MODEL FILES\nVER = 1\n\n# TRAIN RANDOM SEED\nSEED = 42\n\n# FILL NAN VALUE\nNAN_VALUE = -127 # will fit in int8\n\n# FOLDS PER MODEL\nFOLDS = 5","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:22.931019Z","iopub.execute_input":"2025-04-24T18:39:22.931403Z","iopub.status.idle":"2025-04-24T18:39:22.935231Z","shell.execute_reply.started":"2025-04-24T18:39:22.931372Z","shell.execute_reply":"2025-04-24T18:39:22.934566Z"},"trusted":true},"outputs":[],"execution_count":2},{"cell_type":"markdown","source":"# Process and Feature Engineer Train Data\nWe will load @raddar Kaggle dataset from [here][1] with discussion [here][2]. Then we will engineer features suggested by @huseyincot in his notebooks [here][3] and [here][4]. We will use [RAPIDS][5] and the GPU to create new features quickly.\n\n[1]: https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format\n[2]: https://www.kaggle.com/competitions/amex-default-prediction/discussion/328514\n[3]: https://www.kaggle.com/code/huseyincot/amex-catboost-0-793\n[4]: https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n[5]: https://rapids.ai/","metadata":{}},{"cell_type":"code","source":"def check_gpu_memory(status = None):\n    \n    context = cuda.current_context()\n    free_memory, total_memory = context.get_memory_info()\n    \n    print(status)\n    print(f'Total memeory{total_memory}')\n    print(f'Free memeory{free_memory}')\n    ","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:22.93587Z","iopub.execute_input":"2025-04-24T18:39:22.936112Z","iopub.status.idle":"2025-04-24T18:39:22.948493Z","shell.execute_reply.started":"2025-04-24T18:39:22.936089Z","shell.execute_reply":"2025-04-24T18:39:22.947783Z"},"trusted":true},"outputs":[],"execution_count":3},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n    else: df = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n    df = df.sort_values(['customer_ID','S_2'])\n    df = df.reset_index(drop=True)\n    # FILL NAN\n    df = df.fillna(NAN_VALUE) \n    print('shape of data:', df.shape)\n    \n    return df\n\nprint('Reading train data...')\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\ntrain_raw = read_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:22.949201Z","iopub.execute_input":"2025-04-24T18:39:22.949432Z","iopub.status.idle":"2025-04-24T18:39:29.931754Z","shell.execute_reply.started":"2025-04-24T18:39:22.949411Z","shell.execute_reply":"2025-04-24T18:39:29.931106Z"},"trusted":true},"outputs":[{"name":"stdout","text":"Reading train data...\nshape of data: (5531451, 190)\n","output_type":"stream"}],"execution_count":4},{"cell_type":"code","source":"def process_and_feature_engineer(df):\n    \n    gc.collect()\n    torch.cuda.empty_cache()\n    \n    #check_gpu_memory('start')\n    \n    # FEATURE ENGINEERING FROM \n    # https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_features]\n    num_diff_features = [ col + '_diff' for col in num_features]\n    \n    test_num_group = df.groupby(\"customer_ID\")\n    \n    #test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg = test_num_group[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n    \n    #Dataframe for diff features\n    df_diff = df[['customer_ID','S_2']].copy()\n    \n    for col in num_features:\n        test_num_agg[f'{col}_max_min_diff'] = test_num_agg[f'{col}_max']-test_num_agg[f'{col}_min']\n        test_num_agg[f'{col}_last_mean_diff'] = (test_num_agg[f'{col}_last']-test_num_agg[f'{col}_mean']).astype('float32')\n        test_num_agg[f'{col}_last_mean_ratio'] = (test_num_agg[f'{col}_last']/test_num_agg[f'{col}_mean']).astype('float32')\n        test_num_agg[f'{col}_min_max_ratio'] = (test_num_agg[f'{col}_min']/test_num_agg[f'{col}_max']).astype('float32')\n        \n        #Extract diff features \n        new_col=col+'_diff'\n        df_diff[new_col]= test_num_group[col].diff().fillna(0)\n\n    test_cat_agg = test_num_group[cat_features].agg(['count', 'last', 'nunique'])\n    #test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n    \n    df = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    \n    #check_gpu_memory('after agg')\n    \n    del test_num_agg,test_cat_agg,test_num_group\n    print(df.info())\n    gc.collect()\n    torch.cuda.empty_cache()\n    torch.cuda.synchronize()\n    #check_gpu_memory('after del test_num_agg,test_cat_agg')\n        \n    test_num_diff_agg = df_diff.groupby('customer_ID')[num_diff_features].agg(['max'])\n    #test_num_diff_agg = df.groupby(\"customer_ID\")[num_diff_features].agg(['mean', 'std','max'])\n    test_num_diff_agg.columns = ['_'.join(x) for x in test_num_diff_agg.columns]\n        \n\n    df = cudf.concat([df, test_num_diff_agg], axis=1)\n    \n    #new\n    #df = df.fillna(NAN_VALUE)\n    #df = df.replace(np.Inf,128)\n    #df = df.replace(-np.Inf,-127)\n    \n    del test_num_diff_agg\n    print('shape after engineering', df.shape )\n    \n    return df\n\ntrain = process_and_feature_engineer(train_raw)\ndel train_raw","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:29.933757Z","iopub.execute_input":"2025-04-24T18:39:29.933981Z","iopub.status.idle":"2025-04-24T18:39:42.977722Z","shell.execute_reply.started":"2025-04-24T18:39:29.933963Z","shell.execute_reply":"2025-04-24T18:39:42.977069Z"},"trusted":true},"outputs":[{"name":"stdout","text":"<class 'cudf.core.dataframe.DataFrame'>\nIndex: 458913 entries, -9223358381327749917 to 9223350112805974911\nColumns: 1626 entries, P_2_mean to D_68_nunique\ndtypes: float32(903), float64(354), int16(36), int32(11), int64(11), int8(311)\nmemory usage: 3.0 GB\nNone\nshape after engineering (458913, 1803)\n","output_type":"stream"}],"execution_count":5},{"cell_type":"code","source":"#train.isna().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:42.978388Z","iopub.execute_input":"2025-04-24T18:39:42.978661Z","iopub.status.idle":"2025-04-24T18:39:42.982335Z","shell.execute_reply.started":"2025-04-24T18:39:42.978638Z","shell.execute_reply":"2025-04-24T18:39:42.981576Z"},"trusted":true},"outputs":[],"execution_count":6},{"cell_type":"code","source":"# ADD TARGETS\ntargets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\ntargets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\ntargets = targets.set_index('customer_ID')\ntrain = train.merge(targets, left_index=True, right_index=True, how='left')\ntrain.target = train.target.astype('int8')\ndel targets\n\n# NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\ntrain = train.sort_index().reset_index()\n\n# FEATURES\nFEATURES = train.columns[1:-1]\nprint(f'There are {len(FEATURES)} features!')","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:42.983065Z","iopub.execute_input":"2025-04-24T18:39:42.983304Z","iopub.status.idle":"2025-04-24T18:39:44.096051Z","shell.execute_reply.started":"2025-04-24T18:39:42.983283Z","shell.execute_reply":"2025-04-24T18:39:44.095265Z"},"trusted":true},"outputs":[{"name":"stdout","text":"There are 1803 features!\n","output_type":"stream"}],"execution_count":7},{"cell_type":"markdown","source":"# Train XGB\nWe will train using `DeviceQuantileDMatrix`. This has a very small GPU memory footprint.","metadata":{}},{"cell_type":"code","source":"# LOAD XGB LIBRARY\nfrom sklearn.model_selection import KFold\nimport xgboost as xgb\nprint('XGB Version',xgb.__version__)\n\n# NEEDED WITH DeviceQuantileDMatrix BELOW\nclass IterLoadForDMatrix(xgb.core.DataIter):\n    def __init__(self, df=None, features=None, target=None, batch_size=126*1024):\n        self.features = features\n        self.target = target\n        self.df = df\n        self.it = 0 # set iterator to 0\n        self.batch_size = batch_size\n        self.batches = int( np.ceil( len(df) / self.batch_size ) )\n        super().__init__()\n\n    def reset(self):\n        '''Reset the iterator'''\n        self.it = 0\n\n    def next(self, input_data):\n        '''Yield next batch of data.'''\n        if self.it == self.batches:\n            return 0 # Return 0 when there's no more batch.\n        \n        a = self.it * self.batch_size\n        b = min( (self.it + 1) * self.batch_size, len(self.df) )\n        dt = cudf.DataFrame(self.df.iloc[a:b])\n        input_data(data=dt[self.features], label=dt[self.target]) #, weight=dt['weight'])\n        self.it += 1\n        return 1","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:44.09687Z","iopub.execute_input":"2025-04-24T18:39:44.097253Z","iopub.status.idle":"2025-04-24T18:39:44.468756Z","shell.execute_reply.started":"2025-04-24T18:39:44.097225Z","shell.execute_reply":"2025-04-24T18:39:44.468168Z"},"trusted":true},"outputs":[{"name":"stdout","text":"XGB Version 2.0.3\n","output_type":"stream"}],"execution_count":8},{"cell_type":"code","source":"# https://www.kaggle.com/yunchonggan\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/328020\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n    \n    #without name for general  calcaution \n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:44.469382Z","iopub.execute_input":"2025-04-24T18:39:44.46962Z","iopub.status.idle":"2025-04-24T18:39:44.476055Z","shell.execute_reply.started":"2025-04-24T18:39:44.469604Z","shell.execute_reply":"2025-04-24T18:39:44.475363Z"},"trusted":true},"outputs":[],"execution_count":9},{"cell_type":"code","source":"def amex_metric_mod_xgb(y_pred: np.ndarray, dtrain:xgb.DMatrix):\n\n    y_true     = dtrain.get_label()\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    #with name to applied in xgboost\n    return 'Amex_eval', 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:44.476911Z","iopub.execute_input":"2025-04-24T18:39:44.477163Z","iopub.status.idle":"2025-04-24T18:39:44.493233Z","shell.execute_reply.started":"2025-04-24T18:39:44.477133Z","shell.execute_reply":"2025-04-24T18:39:44.492497Z"},"trusted":true},"outputs":[],"execution_count":10},{"cell_type":"code","source":"# XGB MODEL Grid Search PARAMETERS\ngs_xgb_parms = { \n    'max_depth':[3,4,5,6],\n    'gamma': [0.5, 1, 1.5, 2, 5],\n    'learning_rate':[0.05,0.1,0.2,0.3], \n    'subsample':[0.6,0.8,1],\n    'colsample_bytree':[0.6],\n    'objective':['binary:logistic'],\n    'eval_metric':['logloss'],\n    'tree_method':['gpu_hist'],\n    'predictor':['gpu_predictor'],\n    'random_state':[SEED]\n}","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:44.493985Z","iopub.execute_input":"2025-04-24T18:39:44.494173Z","iopub.status.idle":"2025-04-24T18:39:44.508185Z","shell.execute_reply.started":"2025-04-24T18:39:44.49416Z","shell.execute_reply":"2025-04-24T18:39:44.507434Z"},"trusted":true},"outputs":[],"execution_count":11},{"cell_type":"markdown","source":"# Grid Search","metadata":{}},{"cell_type":"raw","source":"\"\"\"\nfrom sklearn.model_selection import KFold,ParameterGrid\n\nbest_score=0\nfor g in ParameterGrid(gs_xgb_parms):\n    importances = []\n    oof = []\n    #train = train.to_pandas() # free GPU memory\n    TRAIN_SUBSAMPLE = 0.1\n    gc.collect()\n\n    skf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\n    for fold,(train_idx, valid_idx) in enumerate(skf.split(\n                train, train.target )):\n\n        # TRAIN WITH SUBSAMPLE OF TRAIN FOLD DATA\n        if TRAIN_SUBSAMPLE<1.0:\n            np.random.seed(SEED)\n            train_idx = np.random.choice(train_idx, \n                       int(len(train_idx)*TRAIN_SUBSAMPLE), replace=False)\n            np.random.seed(None)\n\n        print('#'*25)\n        print('### Fold',fold+1)\n        print('### Train size',len(train_idx),'Valid size',len(valid_idx))\n        print(f'### Training with {int(TRAIN_SUBSAMPLE*100)}% fold data...')\n        print('#'*25)\n\n        # TRAIN, VALID, TEST FOR FOLD K\n        Xy_train = IterLoadForDMatrix(train.loc[train_idx], FEATURES, 'target')\n        X_valid = train.loc[valid_idx, FEATURES]\n        y_valid = train.loc[valid_idx, 'target']\n\n        dtrain = xgb.DeviceQuantileDMatrix(Xy_train, max_bin=256)\n        dvalid = xgb.DMatrix(data=X_valid, label=y_valid)\n\n        # TRAIN MODEL FOLD K\n        model = xgb.train(g, \n                    dtrain=dtrain,\n                    evals=[(dtrain,'train'),(dvalid,'valid')],\n                    num_boost_round=9999,\n                    early_stopping_rounds=100,\n                    verbose_eval=100) \n        model.save_model(f'XGB_v{VER}_fold{fold}.xgb')\n\n        # GET FEATURE IMPORTANCE FOR FOLD K\n        dd = model.get_score(importance_type='weight')\n        df = pd.DataFrame({'feature':dd.keys(),f'importance_{fold}':dd.values()})\n        importances.append(df)\n\n        # INFER OOF FOLD K\n        oof_preds = model.predict(dvalid)\n        print(type(y_valid.values))\n        print(type(oof_preds))\n        acc = amex_metric_mod(y_valid.values.get(), oof_preds)\n        print('Kaggle Metric =',acc,'\\n')\n\n        # SAVE OOF\n        df = train.loc[valid_idx, ['customer_ID','target'] ].copy()\n        df['oof_pred'] = oof_preds\n        oof.append( df )\n        \n        # save if best\n        if acc > best_score:\n            best_score = acc\n            best_grid = g  \n\n        del dtrain, Xy_train, dd, df\n        del X_valid, y_valid, dvalid, model\n        _ = gc.collect()\n        \n#print('#'*25)\n#oof = cudf.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\n#acc = amex_metric_mod(oof.target.values.get(), oof.oof_pred.values.get())\n#print('OVERALL CV Kaggle Metric =',acc)        \n        \n        \n        \nprint (\"OOB: %0.5f\" % best_score )\nprint (\"Grid:\", best_grid)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2025-04-24T17:12:06.757056Z","iopub.execute_input":"2025-04-24T17:12:06.757365Z","iopub.status.idle":"2025-04-24T17:12:06.772146Z","shell.execute_reply.started":"2025-04-24T17:12:06.757346Z","shell.execute_reply":"2025-04-24T17:12:06.771594Z"}}},{"cell_type":"code","source":"# best params after grid search\n# XGB MODEL PARAMETERS\nxgb_parms = { \n    'max_depth':3,\n    'gamma': 0.5,\n    'learning_rate':0.05, \n    'subsample':0.8,\n    'eval_metric':'logloss',\n    'colsample_bytree':0.6, \n    'objective':'binary:logistic',\n    'tree_method':'gpu_hist',\n    'predictor':'gpu_predictor',\n    'random_state':SEED\n}","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:39:44.508813Z","iopub.execute_input":"2025-04-24T18:39:44.508975Z","iopub.status.idle":"2025-04-24T18:39:44.519854Z","shell.execute_reply.started":"2025-04-24T18:39:44.508963Z","shell.execute_reply":"2025-04-24T18:39:44.519263Z"},"trusted":true},"outputs":[],"execution_count":12},{"cell_type":"code","source":"importances = []\noof = []\nTRAIN_SUBSAMPLE = 1.0\ngc.collect()\n\nskf = KFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nfor fold, (train_idx, valid_idx) in enumerate(skf.split(train, train.target)):\n    # TRAIN WITH SUBSAMPLE OF TRAIN FOLD DATA\n    if TRAIN_SUBSAMPLE < 1.0:\n        np.random.seed(SEED)\n        train_idx = np.random.choice(train_idx, \n                       int(len(train_idx) * TRAIN_SUBSAMPLE), replace=False)\n        np.random.seed(None)\n    \n    print('#' * 25)\n    print('### Fold', fold + 1)\n    print('### Train size', len(train_idx), 'Valid size', len(valid_idx))\n    print(f'### Training with {int(TRAIN_SUBSAMPLE * 100)}% fold data...')\n    print('#' * 25)\n    \n    # Free GPU memory\n    gc.collect()\n    torch.cuda.empty_cache()\n    \n    # Process data in smaller batches\n    batch_size = 20000  # Reduced batch size\n    train_batches = [train.loc[train_idx[i:i+batch_size]] for i in range(0, len(train_idx), batch_size)]\n\n    processed_batches = []\n    for batch in train_batches:\n        batch = batch.copy()\n        batch = batch.replace([np.inf, -np.inf], np.nan)\n        batch = batch.fillna(0)\n        processed_batches.append(batch)\n    \n    train_fold = cudf.concat(processed_batches)\n    \n    # Reduce features to lower memory usage\n    selected_features = FEATURES[:30]  # Use only the first 30 features\n    train_fold = train_fold[selected_features + ['target']]\n    X_valid = train.loc[valid_idx, selected_features]\n    y_valid = train.loc[valid_idx, 'target']\n    \n    # Create QuantileDMatrix with CPU if needed\n    xgb_parms['tree_method'] = 'hist'  # Use CPU-based histogram algorithm\n    xgb_parms['predictor'] = 'cpu_predictor'\n    \n    dtrain = xgb.QuantileDMatrix(data=train_fold.to_pandas(), max_bin=128, label=train_fold['target'].to_pandas())  # Reduced max_bin\n    dvalid = xgb.DMatrix(data=X_valid.to_pandas(), label=y_valid.to_pandas())\n    \n    # Train model\n    model = xgb.train(xgb_parms, \n                dtrain=dtrain,\n                evals=[(dtrain, 'train'), (dvalid, 'valid')],\n                num_boost_round=9999,\n                early_stopping_rounds=100,\n                verbose_eval=100) \n    model.save_model(f'XGB_v{VER}_fold{fold}.xgb')\n    \n    # Get feature importance\n    dd = model.get_score(importance_type='weight')\n    df = pd.DataFrame({'feature': dd.keys(), f'importance_{fold}': dd.values()})\n    importances.append(df)\n            \n    # Infer OOF fold\n    oof_preds = model.predict(dvalid)\n    acc = amex_metric_mod(y_valid.values.get(), oof_preds)\n    print('Kaggle Metric =', acc, '\\n')\n    \n    # Save OOF\n    df = train.loc[valid_idx, ['customer_ID', 'target']].copy()\n    df['oof_pred'] = oof_preds\n    oof.append(df)\n    \n    del dtrain, train_fold, dd, df\n    del X_valid, y_valid, dvalid, model\n    _ = gc.collect()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2025-04-24T18:39:44.520606Z","iopub.execute_input":"2025-04-24T18:39:44.52084Z","iopub.status.idle":"2025-04-24T18:43:30.322137Z","shell.execute_reply.started":"2025-04-24T18:39:44.520818Z","shell.execute_reply":"2025-04-24T18:43:30.320148Z"},"trusted":true},"outputs":[{"name":"stdout","text":"#########################\n### Fold 1\n### Train size 367130 Valid size 91783\n### Training with 100% fold data...\n#########################\n","output_type":"stream"},{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mKeyError\u001b[0m                                  Traceback (most recent call last)","\u001b[0;32m/tmp/ipykernel_31/2729891015.py\u001b[0m in \u001b[0;36m<cell line: 0>\u001b[0;34m()\u001b[0m\n\u001b[1;32m     38\u001b[0m     \u001b[0;31m# Reduce features to lower memory usage\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     39\u001b[0m     \u001b[0mselected_features\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mFEATURES\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;36m30\u001b[0m\u001b[0;34m]\u001b[0m  \u001b[0;31m# Use only the first 30 features\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 40\u001b[0;31m     \u001b[0mtrain_fold\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mtrain_fold\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mselected_features\u001b[0m \u001b[0;34m+\u001b[0m \u001b[0;34m[\u001b[0m\u001b[0;34m'target'\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m     41\u001b[0m     \u001b[0mX_valid\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mtrain\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mloc\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mvalid_idx\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mselected_features\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     42\u001b[0m     \u001b[0my_valid\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mtrain\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mloc\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mvalid_idx\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m'target'\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/cudf/utils/performance_tracking.py\u001b[0m in \u001b[0;36mwrapper\u001b[0;34m(*args, **kwargs)\u001b[0m\n\u001b[1;32m     49\u001b[0m                     )\n\u001b[1;32m     50\u001b[0m                 )\n\u001b[0;32m---> 51\u001b[0;31m             \u001b[0;32mreturn\u001b[0m \u001b[0mfunc\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m*\u001b[0m\u001b[0margs\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m**\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m     52\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     53\u001b[0m     \u001b[0;32mreturn\u001b[0m \u001b[0mwrapper\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/cudf/core/dataframe.py\u001b[0m in \u001b[0;36m__getitem__\u001b[0;34m(self, arg)\u001b[0m\n\u001b[1;32m   1382\u001b[0m                 \u001b[0;32mreturn\u001b[0m \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m_apply_boolean_mask\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mBooleanMask\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mmask\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mlen\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mself\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   1383\u001b[0m             \u001b[0;32melse\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m-> 1384\u001b[0;31m                 \u001b[0;32mreturn\u001b[0m \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m_get_columns_by_label\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mmask\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m   1385\u001b[0m         \u001b[0;32melif\u001b[0m \u001b[0misinstance\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0marg\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mDataFrame\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   1386\u001b[0m             \u001b[0;32mreturn\u001b[0m \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mwhere\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0marg\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/cudf/utils/performance_tracking.py\u001b[0m in \u001b[0;36mwrapper\u001b[0;34m(*args, **kwargs)\u001b[0m\n\u001b[1;32m     49\u001b[0m                     )\n\u001b[1;32m     50\u001b[0m                 )\n\u001b[0;32m---> 51\u001b[0;31m             \u001b[0;32mreturn\u001b[0m \u001b[0mfunc\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m*\u001b[0m\u001b[0margs\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m**\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m     52\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     53\u001b[0m     \u001b[0;32mreturn\u001b[0m \u001b[0mwrapper\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/cudf/core/frame.py\u001b[0m in \u001b[0;36m_get_columns_by_label\u001b[0;34m(self, labels)\u001b[0m\n\u001b[1;32m    404\u001b[0m         \u001b[0mAkin\u001b[0m \u001b[0mto\u001b[0m \u001b[0mcudf\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mDataFrame\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m...\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mloc\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mlabels\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    405\u001b[0m         \"\"\"\n\u001b[0;32m--> 406\u001b[0;31m         \u001b[0;32mreturn\u001b[0m \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m_from_data_like_self\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m_data\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mselect_by_label\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mlabels\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    407\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    408\u001b[0m     \u001b[0;34m@\u001b[0m\u001b[0mproperty\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/cudf/core/column_accessor.py\u001b[0m in \u001b[0;36mselect_by_label\u001b[0;34m(self, key)\u001b[0m\n\u001b[1;32m    406\u001b[0m             \u001b[0;32mreturn\u001b[0m \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m_select_by_label_slice\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mkey\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    407\u001b[0m         \u001b[0;32melif\u001b[0m \u001b[0mpd\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mapi\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mtypes\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mis_list_like\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mkey\u001b[0m\u001b[0;34m)\u001b[0m \u001b[0;32mand\u001b[0m \u001b[0;32mnot\u001b[0m \u001b[0misinstance\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mkey\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mtuple\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 408\u001b[0;31m             \u001b[0;32mreturn\u001b[0m \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m_select_by_label_list_like\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mtuple\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mkey\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    409\u001b[0m         \u001b[0;32melse\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    410\u001b[0m             \u001b[0;32mif\u001b[0m \u001b[0misinstance\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mkey\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mtuple\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/cudf/core/column_accessor.py\u001b[0m in \u001b[0;36m_select_by_label_list_like\u001b[0;34m(self, key)\u001b[0m\n\u001b[1;32m    557\u001b[0m             )\n\u001b[1;32m    558\u001b[0m         \u001b[0;32melse\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 559\u001b[0;31m             \u001b[0mdata\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0;34m{\u001b[0m\u001b[0mk\u001b[0m\u001b[0;34m:\u001b[0m \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m_grouped_data\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mk\u001b[0m\u001b[0;34m]\u001b[0m \u001b[0;32mfor\u001b[0m \u001b[0mk\u001b[0m \u001b[0;32min\u001b[0m \u001b[0mkey\u001b[0m\u001b[0;34m}\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    560\u001b[0m             \u001b[0;32mif\u001b[0m \u001b[0mlen\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mdata\u001b[0m\u001b[0;34m)\u001b[0m \u001b[0;34m!=\u001b[0m \u001b[0mlen\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mkey\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    561\u001b[0m                 raise ValueError(\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/cudf/core/column_accessor.py\u001b[0m in \u001b[0;36m<dictcomp>\u001b[0;34m(.0)\u001b[0m\n\u001b[1;32m    557\u001b[0m             )\n\u001b[1;32m    558\u001b[0m         \u001b[0;32melse\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 559\u001b[0;31m             \u001b[0mdata\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0;34m{\u001b[0m\u001b[0mk\u001b[0m\u001b[0;34m:\u001b[0m \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m_grouped_data\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mk\u001b[0m\u001b[0;34m]\u001b[0m \u001b[0;32mfor\u001b[0m \u001b[0mk\u001b[0m \u001b[0;32min\u001b[0m \u001b[0mkey\u001b[0m\u001b[0;34m}\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    560\u001b[0m             \u001b[0;32mif\u001b[0m \u001b[0mlen\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mdata\u001b[0m\u001b[0;34m)\u001b[0m \u001b[0;34m!=\u001b[0m \u001b[0mlen\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mkey\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    561\u001b[0m                 raise ValueError(\n","\u001b[0;31mKeyError\u001b[0m: 'P_2_meantarget'"],"ename":"KeyError","evalue":"'P_2_meantarget'","output_type":"error"}],"execution_count":13},{"cell_type":"code","source":"print('#'*25)\noof = cudf.concat(oof,axis=0,ignore_index=True).set_index('customer_ID')\nacc = amex_metric_mod(oof.target.values.get(), oof.oof_pred.values.get())\nprint('OVERALL CV Kaggle Metric =',acc)","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:43:30.322641Z","iopub.status.idle":"2025-04-24T18:43:30.322926Z","shell.execute_reply.started":"2025-04-24T18:43:30.322751Z","shell.execute_reply":"2025-04-24T18:43:30.32276Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CLEAN RAM\ndel train\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:43:30.324058Z","iopub.status.idle":"2025-04-24T18:43:30.324279Z","shell.execute_reply.started":"2025-04-24T18:43:30.324178Z","shell.execute_reply":"2025-04-24T18:43:30.324188Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save OOF Preds","metadata":{}},{"cell_type":"code","source":"oof=oof.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:43:42.7492Z","iopub.execute_input":"2025-04-24T18:43:42.749942Z","iopub.status.idle":"2025-04-24T18:43:42.762017Z","shell.execute_reply.started":"2025-04-24T18:43:42.749916Z","shell.execute_reply":"2025-04-24T18:43:42.761104Z"},"trusted":true},"outputs":[{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mAttributeError\u001b[0m                            Traceback (most recent call last)","\u001b[0;32m/tmp/ipykernel_31/3646185452.py\u001b[0m in \u001b[0;36m<cell line: 0>\u001b[0;34m()\u001b[0m\n\u001b[0;32m----> 1\u001b[0;31m \u001b[0moof\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0moof\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mto_pandas\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m","\u001b[0;31mAttributeError\u001b[0m: 'list' object has no attribute 'to_pandas'"],"ename":"AttributeError","evalue":"'list' object has no attribute 'to_pandas'","output_type":"error"}],"execution_count":14},{"cell_type":"code","source":"oof_xgb = pd.read_parquet(TRAIN_PATH, columns=['customer_ID']).drop_duplicates()\noof_xgb['customer_ID_hash'] = oof_xgb['customer_ID'].apply(lambda x: int(x[-16:],16) ).astype('int64')\noof_xgb = oof_xgb.set_index('customer_ID_hash')\noof_xgb = oof_xgb.merge(oof, left_index=True, right_index=True)\noof_xgb = oof_xgb.sort_index().reset_index(drop=True)\noof_xgb.to_csv(f'oof_xgb_v{VER}.csv',index=False)\noof_xgb.head()","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:43:30.325999Z","iopub.status.idle":"2025-04-24T18:43:30.3262Z","shell.execute_reply.started":"2025-04-24T18:43:30.326104Z","shell.execute_reply":"2025-04-24T18:43:30.326113Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# PLOT OOF PREDICTIONS\nplt.hist(oof_xgb.oof_pred.values, bins=100)\nplt.title('OOF Predictions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:43:30.327259Z","iopub.status.idle":"2025-04-24T18:43:30.327578Z","shell.execute_reply.started":"2025-04-24T18:43:30.327407Z","shell.execute_reply":"2025-04-24T18:43:30.327421Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CLEAR VRAM, RAM FOR INFERENCE BELOW\ndel oof_xgb, oof\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:43:30.328407Z","iopub.status.idle":"2025-04-24T18:43:30.328649Z","shell.execute_reply.started":"2025-04-24T18:43:30.328551Z","shell.execute_reply":"2025-04-24T18:43:30.328561Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Importance","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndf = importances[0].copy()\nfor k in range(1,FOLDS): df = df.merge(importances[k], on='feature', how='left')\ndf['importance'] = df.iloc[:,1:].mean(axis=1)\ndf = df.sort_values('importance',ascending=False)\ndf.to_csv(f'xgb_feature_importance_v{VER}.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:44:23.073012Z","iopub.execute_input":"2025-04-24T18:44:23.073265Z","iopub.status.idle":"2025-04-24T18:44:23.140506Z","shell.execute_reply.started":"2025-04-24T18:44:23.07324Z","shell.execute_reply":"2025-04-24T18:44:23.139625Z"},"trusted":true},"outputs":[{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mNameError\u001b[0m                                 Traceback (most recent call last)","\u001b[0;32m/tmp/ipykernel_31/2439299887.py\u001b[0m in \u001b[0;36m<cell line: 0>\u001b[0;34m()\u001b[0m\n\u001b[1;32m      1\u001b[0m \u001b[0;32mimport\u001b[0m \u001b[0mmatplotlib\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mpyplot\u001b[0m \u001b[0;32mas\u001b[0m \u001b[0mplt\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m      2\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m----> 3\u001b[0;31m \u001b[0mdf\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mimportances\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;36m0\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mcopy\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m      4\u001b[0m \u001b[0;32mfor\u001b[0m \u001b[0mk\u001b[0m \u001b[0;32min\u001b[0m \u001b[0mrange\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;36m1\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0mFOLDS\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m \u001b[0mdf\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mdf\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mmerge\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mimportances\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mk\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mon\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;34m'feature'\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mhow\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;34m'left'\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m      5\u001b[0m \u001b[0mdf\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m'importance'\u001b[0m\u001b[0;34m]\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mdf\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0miloc\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;36m1\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mmean\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0maxis\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;36m1\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;31mNameError\u001b[0m: name 'importances' is not defined"],"ename":"NameError","evalue":"name 'importances' is not defined","output_type":"error"}],"execution_count":1},{"cell_type":"code","source":"NUM_FEATURES = 50\nplt.figure(figsize=(10,5*NUM_FEATURES//10))\nplt.barh(np.arange(NUM_FEATURES,0,-1), df.importance.values[:NUM_FEATURES])\nplt.yticks(np.arange(NUM_FEATURES,0,-1), df.feature.values[:NUM_FEATURES])\nplt.title(f'XGB Feature Importance - Top {NUM_FEATURES}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:43:30.330383Z","iopub.status.idle":"2025-04-24T18:43:30.330661Z","shell.execute_reply.started":"2025-04-24T18:43:30.330514Z","shell.execute_reply":"2025-04-24T18:43:30.330544Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Process and Feature Engineer Test Data\nWe will load @raddar Kaggle dataset from [here][1] with discussion [here][2]. Then we will engineer features suggested by @huseyincot in his notebooks [here][1] and [here][4]. We will use [RAPIDS][5] and the GPU to create new features quickly.\n\n[1]: https://www.kaggle.com/datasets/raddar/amex-data-integer-dtypes-parquet-format\n[2]: https://www.kaggle.com/competitions/amex-default-prediction/discussion/328514\n[3]: https://www.kaggle.com/code/huseyincot/amex-catboost-0-793\n[4]: https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n[5]: https://rapids.ai/","metadata":{}},{"cell_type":"code","source":"# CALCULATE SIZE OF EACH SEPARATE TEST PART\ndef get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k==NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    if verbose != '': print( rows )\n    return rows,chunk\n\n# COMPUTE SIZE OF 4 PARTS FOR TEST DATA\nNUM_PARTS = 4\nTEST_PATH = '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\nprint(f'Reading test data...')\ntest = read_file(path = TEST_PATH, usecols = ['customer_ID','S_2'])\ncustomers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\nrows,num_cust = get_rows(customers, test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:44:50.551869Z","iopub.execute_input":"2025-04-24T18:44:50.552125Z","iopub.status.idle":"2025-04-24T18:44:50.622427Z","shell.execute_reply.started":"2025-04-24T18:44:50.5521Z","shell.execute_reply":"2025-04-24T18:44:50.621535Z"},"trusted":true},"outputs":[{"name":"stdout","text":"Reading test data...\n","output_type":"stream"},{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mNameError\u001b[0m                                 Traceback (most recent call last)","\u001b[0;32m/tmp/ipykernel_31/2862119458.py\u001b[0m in \u001b[0;36m<cell line: 0>\u001b[0;34m()\u001b[0m\n\u001b[1;32m     21\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     22\u001b[0m \u001b[0mprint\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34mf'Reading test data...'\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 23\u001b[0;31m \u001b[0mtest\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mread_file\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mpath\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mTEST_PATH\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0musecols\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0;34m[\u001b[0m\u001b[0;34m'customer_ID'\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m'S_2'\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m     24\u001b[0m \u001b[0mcustomers\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mtest\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m'customer_ID'\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mdrop_duplicates\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0msort_index\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mvalues\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mflatten\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     25\u001b[0m \u001b[0mrows\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0mnum_cust\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mget_rows\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mcustomers\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mtest\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;34m'customer_ID'\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mNUM_PARTS\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mNUM_PARTS\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mverbose\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0;34m'test'\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;31mNameError\u001b[0m: name 'read_file' is not defined"],"ename":"NameError","evalue":"name 'read_file' is not defined","output_type":"error"}],"execution_count":1},{"cell_type":"markdown","source":"# Infer Test","metadata":{}},{"cell_type":"code","source":"# INFER TEST DATA IN PARTS\nskip_rows = 0\nskip_cust = 0\ntest_preds = []\n\nfor k in range(NUM_PARTS):\n    \n    # READ PART OF TEST DATA\n    print(f'\\nReading test data...')\n    test = read_file(path = TEST_PATH)\n    test = test.iloc[skip_rows:skip_rows+rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test.shape )\n    \n    # PROCESS AND FEATURE ENGINEER PART OF TEST DATA\n    test = process_and_feature_engineer(test)\n    if k==NUM_PARTS-1: test = test.loc[customers[skip_cust:]]\n    else: test = test.loc[customers[skip_cust:skip_cust+num_cust]]\n    skip_cust += num_cust\n    \n    # TEST DATA FOR XGB\n    X_test = test[FEATURES]\n    dtest = xgb.DMatrix(data=X_test)\n    test = test[['P_2_mean']] # reduce memory\n    del X_test\n    gc.collect()\n\n    # INFER XGB MODELS ON TEST DATA\n    model = xgb.Booster()\n    model.load_model(f'XGB_v{VER}_fold0.xgb')\n    preds = model.predict(dtest)\n    for f in range(1,FOLDS):\n        model.load_model(f'XGB_v{VER}_fold{f}.xgb')\n        preds += model.predict(dtest)\n    preds /= FOLDS\n    test_preds.append(preds)\n\n    # CLEAN MEMORY\n    del dtest, model\n    _ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:43:30.332894Z","iopub.status.idle":"2025-04-24T18:43:30.33309Z","shell.execute_reply.started":"2025-04-24T18:43:30.332997Z","shell.execute_reply":"2025-04-24T18:43:30.333006Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create Submission CSV","metadata":{}},{"cell_type":"code","source":"# WRITE SUBMISSION FILE\ntest_preds = np.concatenate(test_preds)\ntest = cudf.DataFrame(index=customers,data={'prediction':test_preds})\nsub = cudf.read_csv('../input/amex-default-prediction/sample_submission.csv')[['customer_ID']]\nsub['customer_ID_hash'] = sub['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\nsub = sub.set_index('customer_ID_hash')\nsub = sub.merge(test[['prediction']], left_index=True, right_index=True, how='left')\nsub = sub.reset_index(drop=True)\n\n# DISPLAY PREDICTIONS\nsub.to_csv(f'submission.csv',index=False)\nprint('Submission file shape is', sub.shape )\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:43:30.334017Z","iopub.status.idle":"2025-04-24T18:43:30.334288Z","shell.execute_reply.started":"2025-04-24T18:43:30.334183Z","shell.execute_reply":"2025-04-24T18:43:30.334195Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# PLOT PREDICTIONS\nplt.hist(sub.to_pandas().prediction, bins=100)\nplt.title('Test Predictions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-04-24T18:43:30.335385Z","iopub.status.idle":"2025-04-24T18:43:30.335703Z","shell.execute_reply.started":"2025-04-24T18:43:30.335548Z","shell.execute_reply":"2025-04-24T18:43:30.335563Z"},"trusted":true},"outputs":[],"execution_count":null}]}