{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport gc\n\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split, StratifiedKFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-15T08:15:32.126591Z","iopub.execute_input":"2022-07-15T08:15:32.127438Z","iopub.status.idle":"2022-07-15T08:15:32.132945Z","shell.execute_reply.started":"2022-07-15T08:15:32.127393Z","shell.execute_reply":"2022-07-15T08:15:32.131696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature engineering\nMoved train & test data processing/feature engineering to this notebook: https://www.kaggle.com/code/canonicalized/amex-v2-feature-files","metadata":{}},{"cell_type":"code","source":"#%%time\ndef read_file(path = '', usecols = None):\n##read a small piece of the parquet file\n#     from pyarrow.parquet import ParquetFile\n#     import pyarrow as pa \n#     pf = ParquetFile(path) \n#     first_ten_rows = next(pf.iter_batches(batch_size = 20))\n#     df = pa.Table.from_batches([first_ten_rows]).to_pandas()\n    df = pd.read_parquet(path, columns=usecols)\n#     df['customer_ID'] = df['customer_ID'].apply(lambda x: int(x[-16:],16) ).astype('int64')\n#     df['S_2'] = pd.to_datetime(df['S_2'])\n    return df\ntrain = read_file('/kaggle/input/amex-v2-feature-files/train_features.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:32:11.527733Z","iopub.execute_input":"2022-07-15T08:32:11.528473Z","iopub.status.idle":"2022-07-15T08:32:15.250559Z","shell.execute_reply.started":"2022-07-15T08:32:11.528431Z","shell.execute_reply":"2022-07-15T08:32:15.249434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import sys\n# np.set_printoptions(threshold=sys.maxsize)\n# train.columns.values","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:15:33.186335Z","iopub.execute_input":"2022-07-15T08:15:33.186683Z","iopub.status.idle":"2022-07-15T08:15:33.190271Z","shell.execute_reply.started":"2022-07-15T08:15:33.186629Z","shell.execute_reply":"2022-07-15T08:15:33.189625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_targets(df):\n    targets = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\n    targets['customer_ID'] = targets['customer_ID'].apply(lambda x: int(x[-16:],16) ).astype('int64')\n    targets['target'] = targets['target'].astype('int8')\n    targets = targets.set_index('customer_ID')\n    df = df.merge(targets, left_index=True, right_index=True, how='left')\n    df.reset_index(inplace=True)\n    del targets\n    return df\n    \ntrain = add_targets(train)\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:15:33.191522Z","iopub.execute_input":"2022-07-15T08:15:33.192167Z","iopub.status.idle":"2022-07-15T08:15:34.495843Z","shell.execute_reply.started":"2022-07-15T08:15:33.192131Z","shell.execute_reply":"2022-07-15T08:15:34.494554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = train.columns[1:-1]\nprint(f'There are {len(features)} features!')\n_ = gc.collect()\n# features.value_counts()\n#pd.DataFrame(features.values).to_csv('features.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:15:34.498575Z","iopub.execute_input":"2022-07-15T08:15:34.499118Z","iopub.status.idle":"2022-07-15T08:15:34.622214Z","shell.execute_reply.started":"2022-07-15T08:15:34.499080Z","shell.execute_reply":"2022-07-15T08:15:34.620949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)\n\n\ndef xgb_amex(y_pred, y_true):\n    return 'amex', amex_metric_np(y_pred,y_true.get_label())\n\n# Created by https://www.kaggle.com/yunchonggan\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/328020\ndef amex_metric_np(preds: np.ndarray, target: np.ndarray) -> float:\n    indices = np.argsort(preds)[::-1]\n    preds, target = preds[indices], target[indices]\n\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n    d = np.sum(target[four_pct_mask]) / np.sum(target)\n\n    weighted_target = target * weight\n    lorentz = (weighted_target / weighted_target.sum()).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    n_pos = np.sum(target)\n    n_neg = target.shape[0] - n_pos\n    gini_max = 10 * n_neg * (n_pos + 20 * n_neg - 19) / (n_pos + 20 * n_neg)\n\n    g = gini / gini_max\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:15:34.623549Z","iopub.execute_input":"2022-07-15T08:15:34.623968Z","iopub.status.idle":"2022-07-15T08:15:34.641665Z","shell.execute_reply.started":"2022-07-15T08:15:34.623933Z","shell.execute_reply":"2022-07-15T08:15:34.640676Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_params = { \n#    'random_state':42,\n#    'booster': 'dart',\n    'max_depth':7,#4, \n    'learning_rate':0.04,#0.03,#0.05,#0.2 \n    'subsample':0.88,#0.8,#0.9,#0.8,\n    'colsample_bytree':0.5,#0.6,#0.5,#0.6, \n    'eval_metric':'logloss',\n    'nthread':4,\n    'tree_method':'hist',#'gpu_hist',\n\n            'gamma':1.5,\n            'min_child_weight':8,\n            'lambda':70,\n            'eta':0.03,\n\n#     'predictor':'gpu_predictor',\n #   'random_state':42,\n#mine\n    'objective':'binary:logistic',\n#     'gamma':0.25,\n#     'min_child_weight':1,\n#     'reg_lambda':10,\n#     'scale_pos_weight':3,\n    'n_estimators': 9999,\n    'early_stopping_rounds': 500\n}\n\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:15:34.642791Z","iopub.execute_input":"2022-07-15T08:15:34.643107Z","iopub.status.idle":"2022-07-15T08:15:34.773585Z","shell.execute_reply.started":"2022-07-15T08:15:34.643075Z","shell.execute_reply":"2022-07-15T08:15:34.772709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLDS = 7\nseeds = [42, 52, 62]\nSEEDNR = len(seeds)\n\ndef modelao(train, FOLDS = 5, seed = 42):\n    ams = []\n    # oof = []\n\n    kf = StratifiedKFold(n_splits=FOLDS, random_state=seed, shuffle=True)\n    #for train_index, test_index in kf.split(train[features]):\n    for fold,(train_index, test_index) in enumerate(kf.split(train, train.target)):\n\n        X_train, X_test = train[features].loc[train_index], train[features].loc[test_index]\n        y_train, y_test = train['target'][train_index], train['target'][test_index]\n\n        clf = xgb.XGBClassifier(**xgb_params, feval=xgb_amex)\n     #     clf = XGBClassifier(seed=42,objective='binary:logistic')\n        clf.fit(X_train, y_train, verbose=100, eval_set=[(X_test,y_test)])\n        clf.save_model(f'XGB_fold{fold}_seed{seed}.xgb')\n \n\n        ppreds = clf.predict_proba(X_test)[:, 1]\n       # oof.append( ppreds )\n\n        am = amex_metric(y_test, ppreds)\n        print(\"AMEX Metric Score:\", am)\n        ams.append(am)\n\n        del X_train, y_train#, df\n        del X_test, y_test, clf\n        _ = gc.collect()\n    \n    return np.mean(ams)\n\nprint(\"Start modelling\")\nfor s in seeds:\n    amex = modelao(train, FOLDS, s)\n    print(\"#\"*25)\n    print(\"seed\",s)\n    print(\"Avg. AMEX Metric Score:\", amex)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:15:34.774675Z","iopub.execute_input":"2022-07-15T08:15:34.775439Z","iopub.status.idle":"2022-07-15T08:16:31.526063Z","shell.execute_reply.started":"2022-07-15T08:15:34.775402Z","shell.execute_reply":"2022-07-15T08:16:31.525034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:16:31.527422Z","iopub.execute_input":"2022-07-15T08:16:31.528106Z","iopub.status.idle":"2022-07-15T08:16:31.680089Z","shell.execute_reply.started":"2022-07-15T08:16:31.528062Z","shell.execute_reply":"2022-07-15T08:16:31.678963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test data","metadata":{}},{"cell_type":"code","source":"skip_rows = 0\nskip_cust = 0\ntest_preds = []\nTEST_PATH = '/kaggle/input/amex-v2-feature-files/test_features.parquet'\n\n# READ PART OF TEST DATA\nprint(f'\\nReading test data...')\ntest = read_file(path = TEST_PATH)\nprint('Test has shape', test.shape )\n\nMAIN_TEST_PATH = '/kaggle/input/amex-data-integer-dtypes-parquet-format/test.parquet'\ntc = read_file(path = MAIN_TEST_PATH, usecols = ['customer_ID','S_2'])\ncustomers = tc[['customer_ID']].drop_duplicates().sort_index().values.flatten()\ndel tc\ngc.collect()\n\nmodel = xgb.XGBClassifier()\nmodel.load_model(f'XGB_fold0_seed42.xgb')\npreds = model.predict_proba(test[features])[:, 1]\nfor f in range(0,FOLDS):\n    print(f'')\n    for s in seeds:\n        if f==0 and s==42: break\n        model.load_model(f'XGB_fold{f}_seed{s}.xgb')\n        preds += model.predict_proba(test[features])[:, 1]\npreds /= (FOLDS*SEEDNR)\ntest_preds = preds #test_preds.append(preds)\n\ndel test, model\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:32:19.056025Z","iopub.execute_input":"2022-07-15T08:32:19.056454Z","iopub.status.idle":"2022-07-15T08:34:15.425153Z","shell.execute_reply.started":"2022-07-15T08:32:19.056417Z","shell.execute_reply":"2022-07-15T08:34:15.424023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission CSV","metadata":{}},{"cell_type":"code","source":"#test_preds = np.concatenate(test_preds)\ntest = pd.DataFrame(index=customers,data={'prediction':test_preds})\nsub = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv')[['customer_ID']]\nsub.shape\nsub['customer_ID_hash'] = sub['customer_ID'].apply(lambda x: int(x[-16:],16) ).astype('int64')\nsub = sub.set_index('customer_ID_hash')\nsub = sub.merge(test[['prediction']], left_index=True, right_index=True, how='left')\nsub = sub.reset_index(drop=True)\n\nsub.to_csv(f'submission.csv',index=False)\nprint('Submission file shape is', sub.shape )\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:36:52.708606Z","iopub.execute_input":"2022-07-15T08:36:52.709064Z","iopub.status.idle":"2022-07-15T08:37:00.087149Z","shell.execute_reply.started":"2022-07-15T08:36:52.709027Z","shell.execute_reply":"2022-07-15T08:37:00.086040Z"},"trusted":true},"execution_count":null,"outputs":[]}]}