{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport joblib\nimport random\n# import cudf, cupy\nimport numpy as np\nimport pandas as pd\n\nimport lightgbm as lgb\nimport xgboost as xgb","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:15.610925Z","iopub.execute_input":"2022-09-13T08:10:15.611241Z","iopub.status.idle":"2022-09-13T08:10:15.619042Z","shell.execute_reply.started":"2022-09-13T08:10:15.611212Z","shell.execute_reply":"2022-09-13T08:10:15.618348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED = 2022\n\ndef seed_everything(SEED):\n    random.seed(SEED)\n    np.random.seed(SEED)\n    os.environ['PYTHONHASHSEED'] = str(SEED)","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:15.620207Z","iopub.execute_input":"2022-09-13T08:10:15.6207Z","iopub.status.idle":"2022-09-13T08:10:15.629212Z","shell.execute_reply.started":"2022-09-13T08:10:15.62067Z","shell.execute_reply":"2022-09-13T08:10:15.628205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loading","metadata":{}},{"cell_type":"code","source":"df_main = pd.read_feather('../input/amex-train-dropped-preprocessed-1/train_dropped_preprocessed.feather')","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:15.630488Z","iopub.execute_input":"2022-09-13T08:10:15.630801Z","iopub.status.idle":"2022-09-13T08:10:20.397826Z","shell.execute_reply.started":"2022-09-13T08:10:15.630771Z","shell.execute_reply":"2022-09-13T08:10:20.396672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_main.loc[df_main['split'] == 'train']\n# df_valid = df_main.loc[df_main['split'] == 'test']","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:20.399322Z","iopub.execute_input":"2022-09-13T08:10:20.399751Z","iopub.status.idle":"2022-09-13T08:10:27.285497Z","shell.execute_reply.started":"2022-09-13T08:10:20.399707Z","shell.execute_reply":"2022-09-13T08:10:27.284584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_main","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.286847Z","iopub.execute_input":"2022-09-13T08:10:27.287203Z","iopub.status.idle":"2022-09-13T08:10:27.336677Z","shell.execute_reply.started":"2022-09-13T08:10:27.287153Z","shell.execute_reply":"2022-09-13T08:10:27.335383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.338285Z","iopub.execute_input":"2022-09-13T08:10:27.33948Z","iopub.status.idle":"2022-09-13T08:10:27.462912Z","shell.execute_reply.started":"2022-09-13T08:10:27.339436Z","shell.execute_reply":"2022-09-13T08:10:27.461704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"check = [col for col in df_train.columns if col.startswith('S_')]","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.464437Z","iopub.execute_input":"2022-09-13T08:10:27.464872Z","iopub.status.idle":"2022-09-13T08:10:27.47106Z","shell.execute_reply.started":"2022-09-13T08:10:27.464817Z","shell.execute_reply":"2022-09-13T08:10:27.469952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.47265Z","iopub.execute_input":"2022-09-13T08:10:27.473485Z","iopub.status.idle":"2022-09-13T08:10:27.482803Z","shell.execute_reply.started":"2022-09-13T08:10:27.473444Z","shell.execute_reply":"2022-09-13T08:10:27.48168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.select_dtypes('object').columns","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.484299Z","iopub.execute_input":"2022-09-13T08:10:27.484945Z","iopub.status.idle":"2022-09-13T08:10:27.742414Z","shell.execute_reply.started":"2022-09-13T08:10:27.484904Z","shell.execute_reply":"2022-09-13T08:10:27.741444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_to_drop = ['customer_ID', 'split', 'target']\nfeats_list = list(set(df_train.columns) ^ set(cols_to_drop))","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.744006Z","iopub.execute_input":"2022-09-13T08:10:27.744475Z","iopub.status.idle":"2022-09-13T08:10:27.750085Z","shell.execute_reply.started":"2022-09-13T08:10:27.744432Z","shell.execute_reply":"2022-09-13T08:10:27.74923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(feats_list)","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.751471Z","iopub.execute_input":"2022-09-13T08:10:27.752109Z","iopub.status.idle":"2022-09-13T08:10:27.762537Z","shell.execute_reply.started":"2022-09-13T08:10:27.752076Z","shell.execute_reply":"2022-09-13T08:10:27.761544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class XGBDataLoader(xgb.core.DataIter):\n#     def __init__(self, df=None, features=None, target=None, batch_size=256):\n#         self.feats_list = features\n#         self.target = target\n#         self.df = df\n#         self.step = 0\n#         self.batch_size = batch_size\n#         self.n_steps = int(np.ceil(len(df) / self.batch_size))\n#         super().__init__()\n    \n#     def reset(self):\n#         '''Reset the iterator'''\n#         self.steps = 0\n        \n#     def next(self, input_data):\n#         '''Yield next batch of data.'''\n#         if self.steps == self.n_steps:\n#             return 0 # Return 0 when there's no more batches.\n        \n#         batch_start = self.step * self.batch_size\n#         batch_end = min((self.step + 1) * self.batch_size, len(self.df))\n#         batch_data = cudf.DataFrame(self.df.iloc[batch_start:batch_end])\n        \n#         input_data(data=batch_data[self.feats_list], label = batch_data[self.target])\n#         self.step += 1\n#         return 1","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.763761Z","iopub.execute_input":"2022-09-13T08:10:27.764142Z","iopub.status.idle":"2022-09-13T08:10:27.770401Z","shell.execute_reply.started":"2022-09-13T08:10:27.764111Z","shell.execute_reply":"2022-09-13T08:10:27.769473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"markdown","source":"## Eval Metric","metadata":{}},{"cell_type":"code","source":"# def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n#     def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n#         df = (pd.concat([y_true, y_pred], axis='columns')\n#               .sort_values('prediction', ascending=False))\n#         df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n#         four_pct_cutoff = int(0.04 * df['weight'].sum())\n#         df['weight_cumsum'] = df['weight'].cumsum()\n#         df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        \n#         return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n#     def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n#         df = (pd.concat([y_true, y_pred], axis='columns')\n#               .sort_values('prediction', ascending=False))\n#         df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n#         df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n#         total_pos = (df['target'] * df['weight']).sum()\n#         df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n#         df['lorentz'] = df['cum_pos_found'] / total_pos\n#         df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        \n#         return df['gini'].sum()\n\n#     def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n#         y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        \n#         return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n#     g = normalized_weighted_gini(y_true, y_pred)\n#     d = top_four_percent_captured(y_true, y_pred)\n\n#     return 0.5 * (g + d)\n\ndef lgb_amex_metric(y_pred, train_dataset):\n    y_true = train_dataset.get_label()\n    \n    return 'amex_metric', amex_metric(y_true, y_pred), True","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.771857Z","iopub.execute_input":"2022-09-13T08:10:27.772456Z","iopub.status.idle":"2022-09-13T08:10:27.781431Z","shell.execute_reply.started":"2022-09-13T08:10:27.772424Z","shell.execute_reply":"2022-09-13T08:10:27.780145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.782695Z","iopub.execute_input":"2022-09-13T08:10:27.783253Z","iopub.status.idle":"2022-09-13T08:10:27.798227Z","shell.execute_reply.started":"2022-09-13T08:10:27.783211Z","shell.execute_reply":"2022-09-13T08:10:27.797464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Training","metadata":{}},{"cell_type":"code","source":"# Model configs\nlgb_params = {\n    # Core params\n    'objective': 'binary',\n    'metric': 'binary_logloss',\n#     'boosting': 'dart',\n    'num_leaves': 100,\n    'learning_rate': 0.01,\n    'num_threads': -1,\n    'device_type': 'gpu',\n    'seed': SEED,\n    # Learning Control params\n    'min_data_in_leaf': 40,\n    'feature_fraction': 0.2,\n    'bagging_fraction': 0.5,\n    'bagging_freq': 10,\n    'bagging_seed': SEED,\n    'feature_fraction': 0.2,\n    'feature_fraction_seed': SEED,\n    'lambda_l2': 2,\n    'verbosity': 100\n}","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.799486Z","iopub.execute_input":"2022-09-13T08:10:27.800002Z","iopub.status.idle":"2022-09-13T08:10:27.811265Z","shell.execute_reply.started":"2022-09-13T08:10:27.799962Z","shell.execute_reply":"2022-09-13T08:10:27.810511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # DataLoader for XGBoost\n# train_loader = AMEXDataLoader(df_train, feats_list, 'target')\n\n# train_set = xgb.DeviceQuantileDMatrix(train_loader, max_bin=256)\n# valid_set = xgb.DMatrix(data=df_valid[feat_list], label=df_valid['target'])","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.81247Z","iopub.execute_input":"2022-09-13T08:10:27.812975Z","iopub.status.idle":"2022-09-13T08:10:27.822003Z","shell.execute_reply.started":"2022-09-13T08:10:27.812937Z","shell.execute_reply":"2022-09-13T08:10:27.821235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:27.823308Z","iopub.execute_input":"2022-09-13T08:10:27.82384Z","iopub.status.idle":"2022-09-13T08:10:29.02245Z","shell.execute_reply.started":"2022-09-13T08:10:27.823798Z","shell.execute_reply":"2022-09-13T08:10:29.021346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.round(len(df_train)/100*80)","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:29.029474Z","iopub.execute_input":"2022-09-13T08:10:29.029809Z","iopub.status.idle":"2022-09-13T08:10:29.037479Z","shell.execute_reply.started":"2022-09-13T08:10:29.029778Z","shell.execute_reply":"2022-09-13T08:10:29.036093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_train_indices = np.random.randint(len(df_train.customer_ID.unique()), size=int(np.round(len(df_train.customer_ID.unique())/100*99)))","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:29.038831Z","iopub.execute_input":"2022-09-13T08:10:29.039263Z","iopub.status.idle":"2022-09-13T08:10:30.470185Z","shell.execute_reply.started":"2022-09-13T08:10:29.039232Z","shell.execute_reply":"2022-09-13T08:10:30.469062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.loc[df_train.customer_ID.isin(df_train.customer_ID.unique()[sub_train_indices])]","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:30.471791Z","iopub.execute_input":"2022-09-13T08:10:30.472131Z","iopub.status.idle":"2022-09-13T08:10:35.338473Z","shell.execute_reply.started":"2022-09-13T08:10:30.4721Z","shell.execute_reply":"2022-09-13T08:10:35.337571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:10:35.33972Z","iopub.execute_input":"2022-09-13T08:10:35.340227Z","iopub.status.idle":"2022-09-13T08:10:35.346313Z","shell.execute_reply.started":"2022-09-13T08:10:35.34019Z","shell.execute_reply":"2022-09-13T08:10:35.345393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DATASET CHECKPOINT ---------------------------------------------------------------------","metadata":{}},{"cell_type":"code","source":"# Dataset for LightGBM\ntrain_dataset = lgb.Dataset(df_train[feats_list], df_train['target'])\n# valid_dataset = lgb.Dataset(df_valid[feats_list], df_valid['target'])","metadata":{"execution":{"iopub.status.busy":"2022-09-13T01:37:31.621706Z","iopub.execute_input":"2022-09-13T01:37:31.622427Z","iopub.status.idle":"2022-09-13T01:37:33.177518Z","shell.execute_reply.started":"2022-09-13T01:37:31.622393Z","shell.execute_reply":"2022-09-13T01:37:33.176466Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_train","metadata":{"execution":{"iopub.status.busy":"2022-09-13T01:37:33.179173Z","iopub.execute_input":"2022-09-13T01:37:33.179556Z","iopub.status.idle":"2022-09-13T01:37:33.246311Z","shell.execute_reply.started":"2022-09-13T01:37:33.179519Z","shell.execute_reply":"2022-09-13T01:37:33.24516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-13T01:37:33.248094Z","iopub.execute_input":"2022-09-13T01:37:33.248828Z","iopub.status.idle":"2022-09-13T01:37:33.379003Z","shell.execute_reply.started":"2022-09-13T01:37:33.248793Z","shell.execute_reply":"2022-09-13T01:37:33.378059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Training\nmodel = lgb.train(\n    params = lgb_params,\n    train_set = train_dataset,\n    num_boost_round = 9000,\n#     valid_sets = [train_dataset, valid_dataset],\n    valid_sets = [train_dataset],\n    early_stopping_rounds = 1500,\n    verbose_eval = 500,\n    feval = lgb_amex_metric\n)","metadata":{"execution":{"iopub.status.busy":"2022-09-13T01:37:33.380807Z","iopub.execute_input":"2022-09-13T01:37:33.381608Z","iopub.status.idle":"2022-09-13T05:52:59.961096Z","shell.execute_reply.started":"2022-09-13T01:37:33.381573Z","shell.execute_reply":"2022-09-13T05:52:59.959904Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_model('lgb_174feats_train_0.8079.txt')","metadata":{"execution":{"iopub.status.busy":"2022-09-13T06:35:37.586705Z","iopub.execute_input":"2022-09-13T06:35:37.587075Z","iopub.status.idle":"2022-09-13T06:35:39.572077Z","shell.execute_reply.started":"2022-09-13T06:35:37.587024Z","shell.execute_reply":"2022-09-13T06:35:39.570967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"saved_model = lgb.Booster(model_file='../input/amex-lgbm-174feats-train-08188/lgb_174feats_train_0.8079.txt')","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:12:40.070795Z","iopub.execute_input":"2022-09-13T08:12:40.071221Z","iopub.status.idle":"2022-09-13T08:12:40.739034Z","shell.execute_reply.started":"2022-09-13T08:12:40.071174Z","shell.execute_reply":"2022-09-13T08:12:40.737671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pred = saved_model.predict(df_train[feats_list].iloc[:len(df_train)//2])","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:12:42.543733Z","iopub.execute_input":"2022-09-13T08:12:42.544354Z","iopub.status.idle":"2022-09-13T08:21:59.008615Z","shell.execute_reply.started":"2022-09-13T08:12:42.544314Z","shell.execute_reply":"2022-09-13T08:21:59.00737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pred","metadata":{"execution":{"iopub.status.busy":"2022-09-13T08:27:03.63565Z","iopub.execute_input":"2022-09-13T08:27:03.636067Z","iopub.status.idle":"2022-09-13T08:27:03.644097Z","shell.execute_reply.started":"2022-09-13T08:27:03.636034Z","shell.execute_reply":"2022-09-13T08:27:03.642911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pred","metadata":{"execution":{"iopub.status.busy":"2022-09-13T07:01:07.577448Z","iopub.execute_input":"2022-09-13T07:01:07.579102Z","iopub.status.idle":"2022-09-13T07:01:07.58938Z","shell.execute_reply.started":"2022-09-13T07:01:07.579006Z","shell.execute_reply":"2022-09-13T07:01:07.588424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pred_2 = saved_model.predict(df_train[feats_list].iloc[len(df_train)//2:])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = df_train.target.iloc[:len(df_train)//2]\ny_train_2 = df_train.target.iloc[len(df_train)//2:]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.best_score","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(y_train, np.where(train_pred > 0.5, 0, 1))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(y_train_2, np.where(train_pred_2 > 0.5, 0, 1))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}