{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# AMEX - simple XGBoost baseline model\n\nI wanted to get an initial baseline model / score. This extremely simple XGBoost model is built using:\n- train/test dataframes that have numerical features converted to 16 bits (for compression - customer_ID replaced by integer as well)\n- the only encoding is a labelencoder on the categorical fields (just to make XGBoost happy)\n- no imputation of missing values\n- no feature engineering\n- uses *only* the most recent statement for each customer\n- no hyperparameter tuning\n\nI created test and train datasets that compress features to 16 bit numerics and drop all but the most recent statement for each customer to make things simpler.\n\nNOTE: I trained with a GPU; turn off the 'gpu_hist' in the XGBClassifier if you want to use CPU (I don't know how long it will take)","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\n\nimport pandas as pd\nimport numpy as np\n\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder, LabelEncoder\nfrom sklearn.compose import ColumnTransformer, make_column_selector\nfrom sklearn.model_selection import StratifiedKFold\nfrom xgboost import XGBClassifier\n\npd.set_option('display.max_columns', None)\npd.set_option('display.float_format', '{:.3f}'.format)\n\nRANDOM_STATE = 42\nINPUT_PATH = Path(\"../input/amex-eda\")","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:18:21.083586Z","iopub.execute_input":"2022-06-07T23:18:21.084207Z","iopub.status.idle":"2022-06-07T23:18:21.09173Z","shell.execute_reply.started":"2022-06-07T23:18:21.084168Z","shell.execute_reply":"2022-06-07T23:18:21.090822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load premade train dataset; 16 bit numerics, most recent statements only\n\nNote: the customer_ID has been replaced with integer c_ID field","metadata":{}},{"cell_type":"code","source":"train = pd.read_feather(INPUT_PATH / \"train_16_recent_data.feather\")\ntrain.set_index(['c_ID'], inplace=True)\ntrain.sort_index(inplace=True)\n\nlabels = pd.read_feather(INPUT_PATH / \"train_labels.feather\")\nlabels.set_index('c_ID', inplace=True)\nlabels.sort_index(inplace=True)\n\ndisplay(train.head(2))\ndisplay(labels.head(2))","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:18:21.119027Z","iopub.execute_input":"2022-06-07T23:18:21.11947Z","iopub.status.idle":"2022-06-07T23:18:21.90632Z","shell.execute_reply.started":"2022-06-07T23:18:21.119438Z","shell.execute_reply":"2022-06-07T23:18:21.905556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preprocessing\n\nHere we encode the categoricals with a LabelEncoder.\n\n(I tried to use a sklearn ColumnTransformer, but it blows out all the numerics back to float64, exhausting memory - I didn't want to fight with it anymore).","metadata":{}},{"cell_type":"code","source":"encoders = {}\ncategoricals = ['D_63', 'D_64', 'D_66', 'D_68', 'B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126']\ndef make_x_y(df, labels=None):\n    df = df.sort_index()\n    for col in categoricals:\n        if not col in encoders:\n            le = LabelEncoder()\n            df[f'{col}_enc'] = le.fit_transform(df[col])\n            df.drop(columns=col, inplace=True)\n            encoders[col] = le\n        else:\n            le = encoders[col]\n            df[f'{col}_enc'] = le.transform(df[col])\n            df.drop(columns=col, inplace=True)\n    if not labels is None:\n        labels = labels.sort_index().target\n    return df, labels\n","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:18:21.908014Z","iopub.execute_input":"2022-06-07T23:18:21.908579Z","iopub.status.idle":"2022-06-07T23:18:21.916804Z","shell.execute_reply.started":"2022-06-07T23:18:21.908541Z","shell.execute_reply":"2022-06-07T23:18:21.915946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### AMEX metric\n\n(thanks to https://www.kaggle.com/code/rohanrao/amex-competition-metric-implementations)","metadata":{}},{"cell_type":"code","source":"def amex_metric_numpy(y_true: np.array, y_pred: np.array) -> float:\n    \n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by descring prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:18:21.918109Z","iopub.execute_input":"2022-06-07T23:18:21.918573Z","iopub.status.idle":"2022-06-07T23:18:21.929203Z","shell.execute_reply.started":"2022-06-07T23:18:21.918538Z","shell.execute_reply":"2022-06-07T23:18:21.928428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train a first model\n\nI use 3-fold cross-validation just to sanity check consistency of results. It's rolled out by hand so that I can use the AMEX scorer.","metadata":{}},{"cell_type":"code","source":"X, y = make_x_y(train, labels)\ndisplay(X.head(2))\ndisplay(X.dtypes)\n\nxgb = XGBClassifier(objective='binary:logistic',\n                    random_state=RANDOM_STATE,\n                    tree_method='gpu_hist')\n\nskf = StratifiedKFold(n_splits=3)\nscores = []\nfor train_idx, test_idx in skf.split(X,y):\n    X_train = X.iloc[train_idx]\n    y_train = y.iloc[train_idx]\n    X_test = X.iloc[test_idx]\n    y_test = y.iloc[test_idx]\n\n    xgb.fit(X_train, y_train)\n    probs = xgb.predict_proba(X_test)[:,1]\n    scores.append(amex_metric_numpy(y_test.to_numpy(), probs))\n\nprint(\"Scores: \")\ndisplay(scores)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:18:21.931396Z","iopub.execute_input":"2022-06-07T23:18:21.932336Z","iopub.status.idle":"2022-06-07T23:18:43.779846Z","shell.execute_reply.started":"2022-06-07T23:18:21.932308Z","shell.execute_reply":"2022-06-07T23:18:43.778999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fit full model","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:13:19.540374Z","iopub.execute_input":"2022-06-07T23:13:19.540953Z","iopub.status.idle":"2022-06-07T23:13:19.54646Z","shell.execute_reply.started":"2022-06-07T23:13:19.540885Z","shell.execute_reply":"2022-06-07T23:13:19.545076Z"}}},{"cell_type":"code","source":"xgb = XGBClassifier(objective='binary:logistic',\n                    random_state=RANDOM_STATE,\n                    tree_method='gpu_hist')\nxgb.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:18:43.781312Z","iopub.execute_input":"2022-06-07T23:18:43.781896Z","iopub.status.idle":"2022-06-07T23:18:50.021366Z","shell.execute_reply.started":"2022-06-07T23:18:43.781859Z","shell.execute_reply":"2022-06-07T23:18:50.02056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# try to free up RAM\ndel(X)\ndel(train)","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:18:50.02249Z","iopub.execute_input":"2022-06-07T23:18:50.022866Z","iopub.status.idle":"2022-06-07T23:18:50.035523Z","shell.execute_reply.started":"2022-06-07T23:18:50.022829Z","shell.execute_reply":"2022-06-07T23:18:50.034746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make Predictions\n\nagain, the test file has been converted to 16 bit numerics and contains only the most recent statement for each customer","metadata":{}},{"cell_type":"code","source":"import gc\n\n# try to free up RAM\ngc.collect()\n\ntest = pd.read_feather(INPUT_PATH / \"test_16_recent_data.feather\")\ntest.set_index(['c_ID'], inplace=True)\ntest.sort_index(inplace=True)\n\n# original customer keys (for submission file)\ncust = pd.read_feather(INPUT_PATH / \"test_cust.feather\")\ncust.set_index('c_ID', inplace=True)\ncust.sort_index(inplace=True)\n\ndisplay(test.head(2))\ndisplay(cust.head(2))\n","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:18:50.036843Z","iopub.execute_input":"2022-06-07T23:18:50.037418Z","iopub.status.idle":"2022-06-07T23:18:51.443327Z","shell.execute_reply.started":"2022-06-07T23:18:50.03738Z","shell.execute_reply":"2022-06-07T23:18:51.442606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test, _ = make_x_y(test)\nprobs = xgb.predict_proba(X_test)\nprobs","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:18:51.444549Z","iopub.execute_input":"2022-06-07T23:18:51.44541Z","iopub.status.idle":"2022-06-07T23:19:05.704881Z","shell.execute_reply.started":"2022-06-07T23:18:51.44537Z","shell.execute_reply":"2022-06-07T23:19:05.70426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make Submission\n\nNOTE: join with the saved customer keys from the raw dataset","metadata":{}},{"cell_type":"code","source":"submit = pd.DataFrame(probs[:,1], columns=['prediction'],index=test.index)\nsubmit = submit.join(cust).reset_index().set_index('customer_ID')\ndisplay(submit.head(2))\n\nsubmit['prediction'].to_csv('./submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-07T23:19:05.706511Z","iopub.execute_input":"2022-06-07T23:19:05.706891Z","iopub.status.idle":"2022-06-07T23:19:09.499688Z","shell.execute_reply.started":"2022-06-07T23:19:05.706853Z","shell.execute_reply":"2022-06-07T23:19:09.498829Z"},"trusted":true},"execution_count":null,"outputs":[]}]}