{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport pickle\nimport gc\nimport xgboost as xgb\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-22T14:47:14.044725Z","iopub.execute_input":"2022-06-22T14:47:14.047259Z","iopub.status.idle":"2022-06-22T14:47:14.862351Z","shell.execute_reply.started":"2022-06-22T14:47:14.047099Z","shell.execute_reply":"2022-06-22T14:47:14.860885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_pickle_obj(filename):\n    with open(filename, 'rb') as file:\n        obj = pickle.load(file)\n    return obj","metadata":{"execution":{"iopub.status.busy":"2022-06-22T14:47:14.865639Z","iopub.execute_input":"2022-06-22T14:47:14.866234Z","iopub.status.idle":"2022-06-22T14:47:14.872464Z","shell.execute_reply.started":"2022-06-22T14:47:14.866183Z","shell.execute_reply":"2022-06-22T14:47:14.87164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = pd.read_pickle(\"../input/amex-train-aggregation-dataset/train_feat_df.pkl\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-22T14:47:14.873711Z","iopub.execute_input":"2022-06-22T14:47:14.874901Z","iopub.status.idle":"2022-06-22T14:47:28.657834Z","shell.execute_reply.started":"2022-06-22T14:47:14.874862Z","shell.execute_reply":"2022-06-22T14:47:28.656491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_cols = [colname for colname in train_df.columns if (colname not in ['customer_ID', 'target', 'last3_target']) ]\nprint(len(feat_cols))","metadata":{"execution":{"iopub.status.busy":"2022-06-22T14:47:28.660338Z","iopub.execute_input":"2022-06-22T14:47:28.660885Z","iopub.status.idle":"2022-06-22T14:47:28.667035Z","shell.execute_reply.started":"2022-06-22T14:47:28.660851Z","shell.execute_reply":"2022-06-22T14:47:28.666283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_numeric_feature_threshholds(num_featcols):\n    feat_threshholds={}\n    for colname in feat_cols:\n        s = train_df[colname]\n        s = s[ (s.isna()==False)  & (s!=0)]\n        \n        vmean = np.mean(s)\n        vmin = np.min(s)\n        vmax = np.max(s)\n        v_01 = np.quantile(s, 0.01)\n        v_99 = np.quantile(s, 0.99)\n        \n        \n        feat_threshholds[colname]={}\n        feat_threshholds[colname]['vmin'] = v_01 - 2*np.abs(v_01)\n        feat_threshholds[colname]['vmax'] = v_99 + 2*np.abs(v_99)\n        \n    return feat_threshholds","metadata":{"execution":{"iopub.status.busy":"2022-06-22T14:47:28.668347Z","iopub.execute_input":"2022-06-22T14:47:28.669319Z","iopub.status.idle":"2022-06-22T14:47:28.681849Z","shell.execute_reply.started":"2022-06-22T14:47:28.669281Z","shell.execute_reply":"2022-06-22T14:47:28.680338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#feat_threshholds = get_numeric_feature_threshholds(feat_cols)\n#for colname in feat_cols:\n#    vmin = feat_threshholds[colname]['vmin']\n#    vmax = feat_threshholds[colname]['vmax']\n    #train_df[colname] = np.clip(train_df[colname], vmin, vmax) \n#    break","metadata":{"execution":{"iopub.status.busy":"2022-06-22T14:47:28.684293Z","iopub.execute_input":"2022-06-22T14:47:28.684897Z","iopub.status.idle":"2022-06-22T14:47:28.6961Z","shell.execute_reply.started":"2022-06-22T14:47:28.684846Z","shell.execute_reply":"2022-06-22T14:47:28.694827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_params={\n    'eta': 0.05,\n    'max_depth': 4,\n    'subsample': 0.6,\n    'colsample_bytree': 0.6,\n    'tree_method': 'hist',\n    'objective': 'binary:logistic',\n    'eval_metric': 'logloss',\n    'seed': 44\n}","metadata":{"execution":{"iopub.status.busy":"2022-06-22T14:47:28.697921Z","iopub.execute_input":"2022-06-22T14:47:28.698481Z","iopub.status.idle":"2022-06-22T14:47:28.711566Z","shell.execute_reply.started":"2022-06-22T14:47:28.698449Z","shell.execute_reply":"2022-06-22T14:47:28.70981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def top_4percent(pred_df):\n    df = pred_df.copy()\n    df = df.sort_values('pred', ascending=False)\n    df['weight'] = df['target'].apply(lambda v: 20 if v==0 else 1)\n    four_percent_cutoff = 0.04 * sum(df['weight'])\n    df['weight_cumsum'] = df['weight'].cumsum()\n    df_cutoff = df[df.weight_cumsum <= four_percent_cutoff]\n    \n    return df_cutoff['target'].sum()/df['target'].sum()\n\ndef weighted_gini(pred_df):\n    df = pred_df.copy()\n    df = df.sort_values('pred', ascending=False)\n    df['weight'] = df['target'].apply(lambda v: 20 if v==0 else 1)\n    df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n    total_pos = (df['target'] * df['weight']).sum()\n    df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n    df['lorentz'] = df['cum_pos_found'] / total_pos\n    df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n    return df['gini'].sum()\n\n\ndef normalized_gini(df):\n    df_true=df[['target']].copy()\n    df_true['pred'] = df_true['target'].copy()\n    \n    G = weighted_gini(df)/weighted_gini(df_true)\n    return G","metadata":{"execution":{"iopub.status.busy":"2022-06-22T14:47:28.713585Z","iopub.execute_input":"2022-06-22T14:47:28.714049Z","iopub.status.idle":"2022-06-22T14:47:28.731822Z","shell.execute_reply.started":"2022-06-22T14:47:28.713988Z","shell.execute_reply":"2022-06-22T14:47:28.73056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, random_state=33, shuffle=True)\nfor foldnum, (train_index, test_index) in enumerate(skf.split(train_df.customer_ID, train_df.target)):\n    print(\"Training Fold:\", foldnum)\n    fold_train_df = train_df.iloc[train_index]\n    fold_val_df = train_df.iloc[test_index]\n    \n    dtrain = xgb.DMatrix(fold_train_df[feat_cols], label=fold_train_df.target)\n    deval = xgb.DMatrix(fold_val_df[feat_cols], label=fold_val_df.target)\n    \n    bst_model = xgb.train(xgb_params, dtrain, \n                          1000, \n                          early_stopping_rounds= 20,\n                          evals=[(dtrain,'train'), (deval, 'eval')],\n                          verbose_eval = 50\n                         )\n    \n    \n    bst_model.save_model(\"bst_model_{}\".format(foldnum))\n    preds = bst_model.predict(deval)\n    \n    fold_val_pred = fold_val_df[['target']].copy()\n    fold_val_pred['pred'] = preds\n    fold_val_pred[['target', 'pred']].to_csv(\"preds_fold_{}.csv\".format(foldnum))\n    \n    \n    print()\n    print()\n    print(\"evaluation metrics\")\n    G = normalized_gini(fold_val_pred[['target', 'pred']])\n    D = top_4percent(fold_val_pred[['target', 'pred']])\n    \n    print(\"Gini:{:.4f}\".format(G))\n    print(\"Default Rate:{:.4f}\".format(D))\n    print(\"Evaluation Metric:{:.4f}\".format( (G+D)/2 ))\n    \n    print()\n    print()\n    \n    xgb.plot_importance(bst_model, max_num_features =10)\n    plt.show()\n    print()\n    print()\n    \n    del dtrain\n    del deval\n    del fold_val_pred\n    del bst_model\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-22T14:47:28.733478Z","iopub.execute_input":"2022-06-22T14:47:28.733926Z","iopub.status.idle":"2022-06-22T14:50:21.292857Z","shell.execute_reply.started":"2022-06-22T14:47:28.733891Z","shell.execute_reply":"2022-06-22T14:50:21.291658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-22T14:50:21.297277Z","iopub.execute_input":"2022-06-22T14:50:21.298121Z","iopub.status.idle":"2022-06-22T14:50:21.448893Z","shell.execute_reply.started":"2022-06-22T14:50:21.298056Z","shell.execute_reply":"2022-06-22T14:50:21.447678Z"},"trusted":true},"execution_count":null,"outputs":[]}]}