{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# AMEX - simple XGBoost baseline model\n\nI wanted to get an initial baseline model / score. This extremely simple XGBoost model is built using:\n- train/test dataframes that have numerical features converted to 16 bits (for compression - customer_ID replaced by integer as well)\n- the only encoding is a labelencoder on the categorical fields (just to make XGBoost happy)\n- no imputation of missing values\n- no feature engineering\n- uses *only* the most recent statement for each customer\n- no hyperparameter tuning\n\nI created test and train datasets that compress features to 16 bit numerics and drop all but the most recent statement for each customer to make things simpler.\n\nNOTE: I trained with a GPU; turn off the 'gpu_hist' in the XGBClassifier if you want to use CPU (I don't know how long it will take)","metadata":{}},{"cell_type":"markdown","source":"# I modified the below notebook.\n# That's simple and easy to understand!\n# I appriciate it!\nhttps://www.kaggle.com/code/eduus710/amex-extremely-simple-xgboost-baseline","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\n\nimport pandas as pd\nimport numpy as np\n\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder, LabelEncoder\nfrom sklearn.compose import ColumnTransformer, make_column_selector\nfrom sklearn.model_selection import StratifiedKFold\nfrom xgboost import XGBClassifier\nimport catboost as catb\nfrom lightgbm import LGBMClassifier\nfrom sklearn.ensemble import RandomForestClassifier,VotingClassifier\nfrom sklearn.ensemble import StackingClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.linear_model import LogisticRegression\nimport random as r\n\npd.set_option('display.max_columns', None)\npd.set_option('display.float_format', '{:.3f}'.format)\n\nRANDOM_STATE = r.randint(1,100)\nINPUT_PATH = Path(\"../input/amex-eda\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T21:03:28.436555Z","iopub.execute_input":"2022-07-13T21:03:28.437224Z","iopub.status.idle":"2022-07-13T21:03:32.980567Z","shell.execute_reply.started":"2022-07-13T21:03:28.437142Z","shell.execute_reply":"2022-07-13T21:03:32.979776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load premade train dataset; 16 bit numerics, most recent statements only\n\nNote: the customer_ID has been replaced with integer c_ID field","metadata":{}},{"cell_type":"code","source":"train = pd.read_feather(INPUT_PATH / \"train_16_recent_data.feather\")\ntrain.set_index(['c_ID'], inplace=True)\ntrain.sort_index(inplace=True)\n\nlabels = pd.read_feather(INPUT_PATH / \"train_labels.feather\")\nlabels.set_index('c_ID', inplace=True)\nlabels.sort_index(inplace=True)\n\ndisplay(train.head(2))\ndisplay(labels.head(2))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T21:03:32.982334Z","iopub.execute_input":"2022-07-13T21:03:32.982761Z","iopub.status.idle":"2022-07-13T21:03:35.712193Z","shell.execute_reply.started":"2022-07-13T21:03:32.982725Z","shell.execute_reply":"2022-07-13T21:03:35.711466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preprocessing\n\nHere we encode the categoricals with a LabelEncoder.\n\n(I tried to use a sklearn ColumnTransformer, but it blows out all the numerics back to float64, exhausting memory - I didn't want to fight with it anymore).","metadata":{}},{"cell_type":"code","source":"encoders = {}\ncategoricals = ['D_63', 'D_64', 'D_66', 'D_68', 'B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126']\ndef make_x_y(df, labels=None):\n    df = df.sort_index()\n    for col in categoricals:\n        if not col in encoders:\n            le = LabelEncoder()\n            df[f'{col}_enc'] = le.fit_transform(df[col])\n            df.drop(columns=col, inplace=True)\n            encoders[col] = le\n        else:\n            le = encoders[col]\n            df[f'{col}_enc'] = le.transform(df[col])\n            df.drop(columns=col, inplace=True)\n    if not labels is None:\n        labels = labels.sort_index().target\n    return df, labels\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T21:03:35.713322Z","iopub.execute_input":"2022-07-13T21:03:35.713773Z","iopub.status.idle":"2022-07-13T21:03:35.721703Z","shell.execute_reply.started":"2022-07-13T21:03:35.713736Z","shell.execute_reply":"2022-07-13T21:03:35.720727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### AMEX metric\n\n(thanks to https://www.kaggle.com/code/rohanrao/amex-competition-metric-implementations)","metadata":{}},{"cell_type":"code","source":"def amex_metric_numpy(y_true: np.array, y_pred: np.array) -> float:\n    \n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by descring prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T21:03:35.723020Z","iopub.execute_input":"2022-07-13T21:03:35.723696Z","iopub.status.idle":"2022-07-13T21:03:35.733193Z","shell.execute_reply.started":"2022-07-13T21:03:35.723659Z","shell.execute_reply":"2022-07-13T21:03:35.732409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train a first model\n\nI use 3-fold cross-validation just to sanity check consistency of results. It's rolled out by hand so that I can use the AMEX scorer.","metadata":{}},{"cell_type":"code","source":"lgbm_param = { \n    'objective': 'binary',\n    'metric': 'binary_logloss',\n    'boosting': 'dart',\n    'seed': RANDOM_STATE,\n    'num_leaves': 100,\n    'learning_rate': 0.01,\n    'feature_fraction': 0.1,\n    'bagging_freq': 10,\n    'bagging_fraction': 0.50,\n    'n_jobs': -1,\n    'lambda_l2': 2,\n    'min_data_in_leaf': 40,\n    'device': 'gpu'\n}\n\nrf_param = {\n    'criterion' : 'entropy',\n    'random_state' : RANDOM_STATE\n}\n\nX, y = make_x_y(train, labels)\nX = X.fillna(X.median())\ndisplay(X.head(2))\ndisplay(X.dtypes)\ndisplay(len(X))\n\nlgbm = LGBMClassifier(**lgbm_param)\nrf = RandomForestClassifier(**rf_param)\n# svm = SVC(C=1.0,gamma=0.1)\n\nxgb = XGBClassifier(objective='binary:logistic',\n                    random_state=RANDOM_STATE,\n                    tree_method='gpu_hist')\n\nmod = catb.CatBoostClassifier(iterations=1000,random_state=RANDOM_STATE,task_type='GPU')\n\n# estimators = [\n#     ('lgbm', lgbm),\n#     ('rf', rf)\n# ]\n# clf = StackingClassifier(estimators=estimators, final_estimator=LogisticRegression(),stack_method='predict_proba')\n\nskf = StratifiedKFold(n_splits=5)\nscores = []\nfor train_idx, test_idx in skf.split(X,y):\n    X_train = X.iloc[train_idx]\n    y_train = y.iloc[train_idx]\n    X_test = X.iloc[test_idx]\n    y_test = y.iloc[test_idx]\n\n    xgb.fit(X_train, y_train)\n    mod.fit(X_train, y_train)\n    lgbm.fit(X_train, y_train)\n#     rf.fit(X_train,y_train)\n#     SVC.fit(X_train,y_train)\n#     clf.fit(X_train,y_train)\n    print()\n    \n    probs1 = xgb.predict_proba(X_test)[:,1]\n    probs2 = mod.predict_proba(X_test)[:,1]\n    probs3 = lgbm.predict_proba(X_test)[:,1]\n#     probs4 = rf.predict_proba(X_test)[:,1]\n    probs = (probs1+probs2+probs3)/3.\n#     probs = clf.predict_proba(X_test)[:,1]\n#     probs = lgbm.predict_proba(X_test)[:,1]\n    scores.append(amex_metric_numpy(y_test.to_numpy(), probs))\n\nprint(\"Scores: \")\ndisplay(scores)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T21:03:35.735869Z","iopub.execute_input":"2022-07-13T21:03:35.736675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fit full model","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:13:19.540374Z","iopub.execute_input":"2022-06-07T23:13:19.540953Z","iopub.status.idle":"2022-06-07T23:13:19.54646Z","shell.execute_reply.started":"2022-06-07T23:13:19.540885Z","shell.execute_reply":"2022-06-07T23:13:19.545076Z"}}},{"cell_type":"code","source":"xgb = XGBClassifier(objective='binary:logistic',\n                    random_state=RANDOM_STATE,\n                    tree_method='gpu_hist')\nxgb.fit(X, y)\nmod = catb.CatBoostClassifier(random_state=RANDOM_STATE,task_type='GPU')\nmod.fit(X,y)\nlgbm = LGBMClassifier(**lgbm_param)\nlgbm.fit(X,y)\nrf = RandomForestClassifier(**rf_param)\nrf.fit(X,y)\nestimators = [\n    ('lgbm', lgbm),\n    ('mod', mod),\n    ('XGB',xgb)\n#     ('RF',rf)\n]\nclf = StackingClassifier(estimators=estimators, final_estimator=LogisticRegression(random_state=RANDOM_STATE),stack_method='predict_proba')\nclf.fit(X_train,y_train)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# try to free up RAM\ndel(X)\ndel(train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make Predictions\n\nagain, the test file has been converted to 16 bit numerics and contains only the most recent statement for each customer","metadata":{}},{"cell_type":"code","source":"import gc\n\n# try to free up RAM\ngc.collect()\n\ntest = pd.read_feather(INPUT_PATH / \"test_16_recent_data.feather\")\ntest.set_index(['c_ID'], inplace=True)\ntest.sort_index(inplace=True)\n\n# original customer keys (for submission file)\ncust = pd.read_feather(INPUT_PATH / \"test_cust.feather\")\ncust.set_index('c_ID', inplace=True)\ncust.sort_index(inplace=True)\n\ntest = test.fillna(test.median())\ncust = cust.fillna(cust.median())\ndisplay(test.head(2))\ndisplay(cust.head(2))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test, _ = make_x_y(test)\nprobs1 = xgb.predict_proba(X_test)\ndisplay(probs1)\nprobs2 = mod.predict_proba(X_test)\ndisplay(probs2)\nprobs3 = lgbm.predict_proba(X_test)\ndisplay(probs3)\n# probs4 = rf.predict_proba(X_test)\n# display(probs4)\nprobs = (probs1+probs2+probs3)/3.\ndisplay(probs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs = clf.predict_proba(X_test)\nprobs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make Submission\n\nNOTE: join with the saved customer keys from the raw dataset","metadata":{}},{"cell_type":"code","source":"submit = pd.DataFrame(probs[:,1], columns=['prediction'],index=test.index)\nsubmit = submit.join(cust).reset_index().set_index('customer_ID')\ndisplay(submit.head(2))\n\nsubmit['prediction'].to_csv('./submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}