{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import cudf\nimport cupy\nimport xgboost as xgb\nimport lightgbm as lgb\nimport catboost as cat\nfrom catboost import CatBoostClassifier\nfrom catboost import CatBoostRegressor\nfrom sklearn.linear_model import LogisticRegression\nimport pickle\nimport gc\n\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\ncudf.__version__","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:49:20.921334Z","iopub.execute_input":"2022-08-07T02:49:20.921776Z","iopub.status.idle":"2022-08-07T02:49:26.021385Z","shell.execute_reply.started":"2022-08-07T02:49:20.92169Z","shell.execute_reply":"2022-08-07T02:49:26.020609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGB+LGBM+CatBoost\n\nCompared with the other two models, the accuracy of catboost is slightly lower, about 0.790. If you are interested, you can try to improve the accuracy of catboost. I believe it will be helpful to the final score.","metadata":{}},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"code","source":"def get_not_used():\n    # cid is the label encode of customer_ID\n    # row_id indicates the order of rows\n    misscols= ['D_88','D_110','B_39','D_73','B_42','D_134','B_29','D_76','D_132','D_42','D_142','D_53']\n    skew=['B_31', 'D_87']\n    return ['row_id', 'customer_ID', 'target', 'cid', 'S_2','month']+skew+misscols[:-5]\n    \ndef preprocess(df):\n    df['row_id'] = cupy.arange(df.shape[0])\n    not_used = get_not_used()\n    cat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120',\n                'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\n    for col in df.columns:\n        if col not in not_used+cat_cols:\n            df[col] = df[col].round(2)\n\n    df['S_2'] = cudf.to_datetime(df['S_2'])\n    df['cid'], _ = df.customer_ID.factorize()\n    df= df.sort_values(['customer_ID','S_2'])\n    df['month']= df['S_2'].dt.day\n    df['month'] =df.to_pandas().groupby('customer_ID')['month'].diff()\n    \n    important = ['P_2','B_1','B_4','D_39']\n    num_cols = [col for col in df.columns if col not in cat_cols+not_used]+['month']\n    nth_cols = [col for col in df.columns if col not in cat_cols+not_used][:30]+['customer_ID']\n    \n    dgs = add_stats_step(df, num_cols, cat_cols)\n        \n    # cudf merge changes row orders\n    # restore the original row order by sorting row_id\n    df= df.merge(df[nth_cols].groupby('customer_ID').nth(-2),on='customer_ID',how='left',suffixes=[\"_last_1\",\"_last_2\"])\n    df = df.sort_values('row_id')\n    df = df.drop(['row_id'],axis=1)\n    return df, dgs\n\ndef add_stats_step(df, numcols, catcols):\n    n = 50\n    dgs = []\n    for i in range(0,len(numcols),n):\n        s = i\n        e = min(s+n, len(numcols))\n        dg = add_stats_one_shot_num(df, numcols[s:e])\n        dgs.append(dg)\n    for i in range(0,len(catcols),n):\n        s = i\n        e = min(s+n, len(catcols))\n        dg = add_stats_one_shot_cat(df, catcols[s:e])\n        dgs.append(dg)\n    return dgs\n\ndef add_stats_one_shot_num(df, cols):\n    stats = ['mean','max','min','last']\n    dg = df.groupby('customer_ID').agg({col:stats for col in cols})\n    out_cols = []\n    for col in cols:\n        out_cols.extend([f'{col}_{s}' for s in stats])\n    dg.columns = out_cols\n    dg = dg.reset_index()\n    return dg\n\ndef add_stats_one_shot_cat(df, cols):\n    stats = ['count', 'nunique','mean','last','max','min']\n    dg = df.groupby('customer_ID').agg({col:stats for col in cols})\n    out_cols = []\n    for col in cols:\n        out_cols.extend([f'{col}_{s}' for s in stats])\n    dg.columns = out_cols\n    for col in cols:\n        df[f'{col}_cat_diff'] =df.to_pandas().groupby('customer_ID')[col].diff().iloc[[-1]]\n    for col in cols:\n        df[f'{col}_cat_pct_change'] =df.to_pandas().groupby('customer_ID')[col].pct_change().iloc[[-1]]\n    for col in cols:\n        dg[f'{col}_last_mean'] = dg[f'{col}_last'] - dg[f'{col}_mean']\n    for col in cols:\n        dg[f'{col}_max_min'] = dg[f'{col}_max'] - dg[f'{col}_min']\n    dg = dg.reset_index()\n    return dg\n\ndef load_test_iter(path, chunks=15):\n    \n    test_rows = 11363762\n    chunk_rows = test_rows // chunks\n    \n    test = cudf.read_parquet(f'{path}/test.parquet',\n                             columns=['customer_ID','S_2'],\n                             num_rows=test_rows)\n    test = get_segment(test)\n    start = 0\n    while start < test.shape[0]:\n        if start+chunk_rows < test.shape[0]:\n            end = test['cus_count'].values[start+chunk_rows]\n        else:\n            end = test['cus_count'].values[-1]\n        end = int(end)\n        df = cudf.read_parquet(f'{path}/test.parquet',\n                               num_rows = end-start, skiprows=start)\n        start = end\n        yield process_data(df)\n\ndef load_train(path):\n    train = cudf.read_parquet(f'{path}/train.parquet')\n    \n    train = process_data(train)\n    trainl = cudf.read_csv(f'../input/amex-default-prediction/train_labels.csv')\n    train = train.merge(trainl, on='customer_ID', how='left')\n    return train\n\ndef process_data(df):\n    df,dgs = preprocess(df)\n    df = df.drop_duplicates('customer_ID',keep='last')\n    for dg in dgs:\n        df = df.merge(dg, on='customer_ID', how='left')\n#     diff_cols = [col for col in df.columns if col.endswith('_diff')]\n#     df = df.drop(diff_cols,axis=1)\n    return df\n\n\ndef get_segment(test):\n    dg = test.groupby('customer_ID').agg({'S_2':'count'})\n    dg.columns = ['cus_count']\n    dg = dg.reset_index()\n    dg['cid'],_ = dg['customer_ID'].factorize()\n    dg = dg.sort_values('cid')\n    dg['cus_count'] = dg['cus_count'].cumsum()\n    \n    test = test.merge(dg, on='customer_ID', how='left')\n    test = test.sort_values(['cid','S_2'])\n    assert test['cus_count'].values[-1] == test.shape[0]\n    return test","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:49:26.023255Z","iopub.execute_input":"2022-08-07T02:49:26.023626Z","iopub.status.idle":"2022-08-07T02:49:26.049866Z","shell.execute_reply.started":"2022-08-07T02:49:26.023574Z","shell.execute_reply":"2022-08-07T02:49:26.048927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### XGB/LGB Params and utility functions","metadata":{}},{"cell_type":"code","source":"def xgb_train(x, y, xt, yt):\n    print(\"-----------xgb starts training-----------\")\n    print(\"# of features:\", x.shape[1])\n    assert x.shape[1] == xt.shape[1]\n    dtrain = xgb.DMatrix(data=x, label=y)\n    dvalid = xgb.DMatrix(data=xt, label=yt)\n    params = {\n            'objective': 'binary:logistic', \n            'tree_method': 'gpu_hist', \n            'max_depth': 7,\n            'subsample':0.88,\n            'colsample_bytree': 0.5,\n            'gamma':1.5,\n            'min_child_weight':8,\n            'lambda':70,\n            'eta':0.03,\n#             'scale_pos_weight': scale_pos_weight,\n    }\n    watchlist = [(dtrain, 'train'), (dvalid, 'eval')]\n    bst = xgb.train(params, dtrain=dtrain,\n                num_boost_round=20500,evals=watchlist,\n                early_stopping_rounds=500, feval=xgb_amex, maximize=True,\n                verbose_eval=100)\n    print('best ntree_limit:', bst.best_ntree_limit)\n    print('best score:', bst.best_score)\n    return bst.predict(dtrain, iteration_range=(0,bst.best_ntree_limit)), bst.predict(dvalid, iteration_range=(0,bst.best_ntree_limit)), bst","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:49:26.051042Z","iopub.execute_input":"2022-08-07T02:49:26.052065Z","iopub.status.idle":"2022-08-07T02:49:26.067982Z","shell.execute_reply.started":"2022-08-07T02:49:26.05203Z","shell.execute_reply":"2022-08-07T02:49:26.066994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_train(x, y, xt, yt):\n    print(\"----------lgb starts training----------\")\n    print(\"# of features:\", x.shape[1])\n    assert x.shape[1] == xt.shape[1]\n    lgb_train = lgb.Dataset(x.to_pandas(), y.to_pandas())\n    lgb_eval = lgb.Dataset(xt.to_pandas(), yt.to_pandas(), reference=lgb_train)\n    params = {\n        'objective': 'binary',\n        'metric': 'binary_logloss',\n        'boosting':'gbdt',\n        'seed': 42,\n        'num_leaves': 100,\n        'learning_rate': 0.01,\n        'feature_fraction': 0.20,\n        'bagging_freq': 10,\n        'bagging_fraction': 0.50,\n        'n_jobs': -1,\n        'lambda_l2': 2,\n        'min_data_in_leaf': 40\n       \n    }\n    gbm = lgb.train(params,\n                lgb_train,\n                num_boost_round=20500,\n                valid_sets=[lgb_train, lgb_eval],\n                early_stopping_rounds=500,feval=amex_metric_mod_lgbm, \n                verbose_eval=100,)\n\n\n    print('best iterations:', gbm.best_iteration)\n    print('best score:', gbm.best_score)\n    return gbm.predict(x.to_pandas(), num_iteration =gbm.best_iteration),gbm.predict(xt.to_pandas(), num_iteration =gbm.best_iteration), gbm","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:49:26.07043Z","iopub.execute_input":"2022-08-07T02:49:26.070962Z","iopub.status.idle":"2022-08-07T02:49:26.081443Z","shell.execute_reply.started":"2022-08-07T02:49:26.070928Z","shell.execute_reply":"2022-08-07T02:49:26.080644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cat_train(x, y, xt, yt):\n    print(\"-----------catboost starts training-----------\")\n    print(\"# of features:\", x.shape[1])\n    assert x.shape[1] == xt.shape[1]\n    cat_train = cat.Pool(x.to_pandas(), y.to_pandas())\n    cat_eval = cat.Pool(xt.to_pandas(), yt.to_pandas())\n    \n    clf = CatBoostRegressor(iterations=3000, \n                             task_type='GPU',\n                             bagging_temperature = 0.2,\n                             od_type='Iter',\n                             metric_period = 50,\n                             od_wait=20)\n    clf.fit(cat_train, eval_set=cat_eval, verbose=100,early_stopping_rounds=500)\n    return  clf.predict(cat_eval), clf","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:49:26.082569Z","iopub.execute_input":"2022-08-07T02:49:26.083145Z","iopub.status.idle":"2022-08-07T02:49:26.094902Z","shell.execute_reply.started":"2022-08-07T02:49:26.083111Z","shell.execute_reply":"2022-08-07T02:49:26.093981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Metrics","metadata":{}},{"cell_type":"code","source":"def lgb_amex(y_pred, y_true):\n    return 'amex', amex_metric_np(y_pred,y_true.get_label()), True\n\ndef xgb_amex(y_pred, y_true):\n    return 'amex', amex_metric_np(y_pred,y_true.get_label())\n\n# Created by https://www.kaggle.com/yunchonggan\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/328020\ndef amex_metric_np(preds: np.ndarray, target: np.ndarray) -> float:\n    indices = np.argsort(preds)[::-1]\n    preds, target = preds[indices], target[indices]\n\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n    d = np.sum(target[four_pct_mask]) / np.sum(target)\n\n    weighted_target = target * weight\n    lorentz = (weighted_target / weighted_target.sum()).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    n_pos = np.sum(target)\n    n_neg = target.shape[0] - n_pos\n    gini_max = 10 * n_neg * (n_pos + 20 * n_neg - 19) / (n_pos + 20 * n_neg)\n\n    g = gini / gini_max\n    return 0.5 * (g + d)\n\n# we still need the official metric since the faster version above is slightly off\ndef amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)\n\ndef amex_metric_mod_lgbm(y_pred: np.ndarray, data: lgb.Dataset):\n\n    y_true = data.get_label()\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 'AMEX', 0.5 * (gini[1]/gini[0]+ top_four), True","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:49:26.097941Z","iopub.execute_input":"2022-08-07T02:49:26.098249Z","iopub.status.idle":"2022-08-07T02:49:26.11759Z","shell.execute_reply.started":"2022-08-07T02:49:26.098223Z","shell.execute_reply":"2022-08-07T02:49:26.116473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load data and add feature","metadata":{}},{"cell_type":"code","source":"%%time\n#(458913, 1123)(458913, 1156)\npath = '../input/amex-data-integer-dtypes-parquet-format'\ntrain = load_train(path)\n# train2 = load_train(path,2)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-07T02:49:26.118999Z","iopub.execute_input":"2022-08-07T02:49:26.119579Z","iopub.status.idle":"2022-08-07T03:08:55.438885Z","shell.execute_reply.started":"2022-08-07T02:49:26.119544Z","shell.execute_reply":"2022-08-07T03:08:55.438001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head","metadata":{"execution":{"iopub.status.busy":"2022-08-07T03:08:55.440195Z","iopub.execute_input":"2022-08-07T03:08:55.441051Z","iopub.status.idle":"2022-08-07T03:08:55.780175Z","shell.execute_reply.started":"2022-08-07T03:08:55.441012Z","shell.execute_reply":"2022-08-07T03:08:55.779395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train Stacking Model in K-folds","metadata":{}},{"cell_type":"code","source":"%%time\n\nnot_used = get_not_used()\nnot_used = [i for i in not_used if i in train.columns]\nmsgs = {}\nfolds = 5\nscore = 0\n\n#set diff cols if u wanna try different features on xgb & lgbm\n#diff_cols= [col for col in train.columns if col.endswith('_max') or col.endswith('_min') or col.endswith('_mean') or col.endswith('_std') or col.endswith('_count')]\n\n\nfor i in range(folds):\n    print(f\"==============Folds {i}===============\")\n    mask = train['cid']%folds == i\n    tr,va = train[~mask], train[mask]\n\n    x, y = tr.drop(not_used, axis=1), tr['target']\n    xt, yt = va.drop(not_used, axis=1), va['target']\n    \n    xp, yp, bst = xgb_train(x, y, xt, yt)\n    bst.save_model(f'xgb_{i}.json')\n\n#     x = tr.drop(not_used+diff_cols, axis=1)\n#     xt = va.drop(not_used+diff_cols, axis=1)\n    \n    xp2,yp2,gbm = lgb_train(x, y, xt, yt)\n    gbm.save_model(f'lgb_{i}.json')\n    \n    yp3,cats = cat_train(x, y, xt, yt)\n    cats.save_model(f'cat_{i}.json')\n\n#     X_estimator = np.concatenate((np.expand_dims(yp, 1), np.expand_dims(yp2, 1)), 1)\n#     final_estimator = LogisticRegression(penalty=\"l1\",solver=\"saga\",C=0.001,max_iter=1000,class_weight=\"balanced\")\n#     final_estimator.fit(X_estimator, yt.to_pandas())\n#     with open(f'lr_{i}.pkl','wb') as f:\n#         pickle.dump(final_estimator, f)\n#     preds=final_estimator.predict_proba(np.concatenate((np.expand_dims(yp, 1), np.expand_dims(yp2, 1)), 1))[:,1]\n    preds = yp * 0.45+yp2 * 0.45+yp3 * 0.1\n#    preds = yp3\n    amex_score = amex_metric(pd.DataFrame({'target':yt.values.get()}), \n                                    pd.DataFrame({'prediction':preds}))\n    msg = f\"Fold {i} amex {amex_score:.4f}\"\n    print(msg)\n    score += amex_score\n    del tr,va,x,y\n    del xt,yt,bst,gbm\n    _ = gc.collect()\nscore /= folds\nprint(f\"Average amex score: {score:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T03:08:55.781484Z","iopub.execute_input":"2022-08-07T03:08:55.782065Z","iopub.status.idle":"2022-08-07T07:53:26.110262Z","shell.execute_reply.started":"2022-08-07T03:08:55.782007Z","shell.execute_reply":"2022-08-07T07:53:26.10947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:53:26.112719Z","iopub.execute_input":"2022-08-07T07:53:26.113632Z","iopub.status.idle":"2022-08-07T07:53:26.234297Z","shell.execute_reply.started":"2022-08-07T07:53:26.113578Z","shell.execute_reply":"2022-08-07T07:53:26.233273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Inference","metadata":{}},{"cell_type":"code","source":"%%time\ncids = []\nyps = []\n# set chunks \nchunks = 15\n\n\nfor df in tqdm(load_test_iter(path,chunks),total=chunks):\n    cids.append(df['customer_ID'])\n    not_used = [i for i in not_used if i in df.columns]\n\n    preds=0\n    for i in range(folds):\n        bst = xgb.Booster()\n        bst.load_model(f'xgb_{i}.json')\n        dx = xgb.DMatrix(df.drop(not_used, axis=1))\n        \n        gbm = lgb.Booster(model_file=f'lgb_{i}.json')\n        dx2 = df.drop(not_used, axis=1).to_pandas()\n        \n        cats.load_model(f'cat_{i}.json')\n#         with open(f'lr_{i}.pkl','rb') as f:\n#             Lr= pickle.load(f)\n        \n        yp = bst.predict(dx, iteration_range=(0,bst.best_ntree_limit))\n        yp2 = gbm.predict(dx2, num_iteration =gbm.best_iteration)\n        yp3 = cats.predict(dx2)\n        #preds+=final_estimator.predict_proba(np.concatenate((np.expand_dims(yp, 1), np.expand_dims(yp2, 1)), 1))[:,1]\n        preds+=(yp * 0.45+yp2 * 0.45+yp3 * 0.1)\n    yps.append(preds/folds)\n    \ndf = cudf.DataFrame()\ndf['customer_ID'] = cudf.concat(cids)\ndf['prediction'] = np.concatenate(yps)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T07:53:26.235449Z","iopub.execute_input":"2022-08-07T07:53:26.236277Z","iopub.status.idle":"2022-08-07T09:07:23.292813Z","shell.execute_reply.started":"2022-08-07T07:53:26.23624Z","shell.execute_reply":"2022-08-07T09:07:23.291861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T09:07:38.456862Z","iopub.execute_input":"2022-08-07T09:07:38.457974Z","iopub.status.idle":"2022-08-07T09:07:38.611937Z","shell.execute_reply.started":"2022-08-07T09:07:38.457923Z","shell.execute_reply":"2022-08-07T09:07:38.61105Z"},"trusted":true},"execution_count":null,"outputs":[]}]}